From 9417e379f96ed427147b91846b65614f9aa5c9dc Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Tue, 29 Sep 2026 16:40:12 +0530 Subject: [PATCH 01/12] fix(fees): route platform fee payouts to routstr-fees@rizful.com Replace the hard-coded npub.cash Lightning address used by the 2.1% platform fee payout with routstr-fees@rizful.com. --- routstr/auth.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/routstr/auth.py b/routstr/auth.py index bf29e86e..f3624ef4 100644 --- a/routstr/auth.py +++ b/routstr/auth.py @@ -49,9 +49,7 @@ payments_logger = get_logger("routstr.payments") # Routstr platform fee constants ROUTSTR_FEE_PERCENT: float = 2.1 -ROUTSTR_LN_ADDRESS: str = ( - "npub130mznv74rxs032peqym6g3wqavh472623mt3z5w73xq9r6qqdufs7ql29s@npub.cash" -) +ROUTSTR_LN_ADDRESS: str = "routstr-fees@rizful.com" ROUTSTR_FEE_PAYOUT_INTERVAL_SECONDS: int = 900 ROUTSTR_FEE_DEFAULT_PAYOUT: int = 200 From 8239b07cef51c04007cdb52c9a575340fec80b91 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 00:13:06 +0200 Subject: [PATCH 02/12] chore: bump litellm to 1.101.2 for gpt-6 max_completion_tokens --- pyproject.toml | 2 +- tests/unit/test_litellm_routing.py | 21 ++++++ uv.lock | 104 ++++++++++++++++++++--------- 3 files changed, 95 insertions(+), 32 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index c723ae37..296fab3f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -21,7 +21,7 @@ dependencies = [ "mdurl==0.1.2", "pillow>=10", "openai>=1.98.0", - "litellm>=1.93.0,<1.94", # 1.93 is the first line supporting Python 3.14 + "litellm>=1.101.2,<1.102", "orjson>=3.10", ] diff --git a/tests/unit/test_litellm_routing.py b/tests/unit/test_litellm_routing.py index 264a142d..077a765b 100644 --- a/tests/unit/test_litellm_routing.py +++ b/tests/unit/test_litellm_routing.py @@ -91,3 +91,24 @@ def test_detect_litellm_prefix_custom_default() -> None: assert detect_litellm_prefix("https://example.com", default="anthropic/") == ( "anthropic/" ) + + +@pytest.mark.parametrize("model", ["gpt-6", "gpt-6-luna", "gpt-5.5"]) +def test_litellm_sends_max_completion_tokens_for_gpt_5_and_later(model: str) -> None: + """OpenAI rejects ``max_tokens`` on these models; litellm <1.101 only + rewrote it for names containing ``gpt-5``, so gpt-6 got a 400.""" + import litellm + from litellm.utils import ProviderConfigManager + + config = ProviderConfigManager.get_provider_chat_config( + model=model, provider=litellm.LlmProviders.OPENAI + ) + assert config is not None + mapped = config.map_openai_params( + non_default_params={"max_tokens": 10}, + optional_params={}, + model=model, + drop_params=True, + ) + + assert mapped == {"max_completion_tokens": 10} diff --git a/uv.lock b/uv.lock index fad67b51..86d2be5d 100644 --- a/uv.lock +++ b/uv.lock @@ -337,6 +337,34 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9d/9e/78e59887cbf94116bdc890af7726ae264d55df14f1c777724c656e8a35fe/bolt11-2.1.1-py3-none-any.whl", hash = "sha256:fd4edb9e73e27bf5e017f47c97f7c6827b523fcf9cab152b123961ca78323e2d", size = 17102, upload-time = "2025-03-12T13:33:08.142Z" }, ] +[[package]] +name = "boto3" +version = "1.43.105" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "botocore" }, + { name = "jmespath" }, + { name = "s3transfer" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/75/46/d8c87ada70a7647fb3d206c7f19eafca3580a0ae4c06d62da539a1ee1207/boto3-1.43.105.tar.gz", hash = "sha256:e51260aed9cc1474778b5488bc6f97ad28f27a0a7002f4bbaaf8191aff1422ea", size = 112682, upload-time = "2026-09-29T19:37:40.784Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bc/8e/0310a37ff609529dab9153cbc9fd0b66c685364d741bb6d1ae31134b728e/boto3-1.43.105-py3-none-any.whl", hash = "sha256:b8b6236ae7fe2724eee608c9b0649afbb86f0ec98158f39f67e64e678ec47499", size = 140042, upload-time = "2026-09-29T19:37:39.415Z" }, +] + +[[package]] +name = "botocore" +version = "1.43.105" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "jmespath" }, + { name = "python-dateutil" }, + { name = "urllib3" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/2b/30/668f3c0533a440787e212cf56404cb6ec234ae8e6baf97fe17329d512d88/botocore-1.43.105.tar.gz", hash = "sha256:afb3e7706b123ab069d1c34571ca1fdf82528a48425574fe4693df3d039d503f", size = 16263910, upload-time = "2026-09-29T19:37:36.456Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f3/94/50923cd46840e4d2b56cad1dcf5008fb20c099b04b2d93d32231f2d0bfaa/botocore-1.43.105-py3-none-any.whl", hash = "sha256:7abd19e1ef2c5e4a0314ca493fa7cebefabe33e559d7dd570fe2432a5431e6ec", size = 15958067, upload-time = "2026-09-29T19:37:33.373Z" }, +] + [[package]] name = "brotli" version = "1.2.0" @@ -1509,6 +1537,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/b3/4a/4175a563579e884192ba6e81725fc0448b042024419be8d83aa8a80a3f44/jiter-0.10.0-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3aa96f2abba33dc77f79b4cf791840230375f9534e5fac927ccceb58c5e604a5", size = 354213, upload-time = "2025-05-18T19:04:41.894Z" }, ] +[[package]] +name = "jmespath" +version = "1.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d3/59/322338183ecda247fb5d1763a6cbe46eff7222eaeebafd9fa65d4bf5cb11/jmespath-1.1.0.tar.gz", hash = "sha256:472c87d80f36026ae83c6ddd0f1d05d4e510134ed462851fd5f754c8c3cbb88d", size = 27377, upload-time = "2026-01-22T16:35:26.279Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/14/2f/967ba146e6d58cf6a652da73885f52fc68001525b4197effc174321d70b4/jmespath-1.1.0-py3-none-any.whl", hash = "sha256:a5663118de4908c91729bea0acadca56526eb2698e83de10cd116ae0f4e97c64", size = 20419, upload-time = "2026-01-22T16:35:24.919Z" }, +] + [[package]] name = "jsonschema" version = "4.26.0" @@ -1552,10 +1589,11 @@ wheels = [ [[package]] name = "litellm" -version = "1.93.2" +version = "1.101.2" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "aiohttp" }, + { name = "boto3" }, { name = "click" }, { name = "fastuuid" }, { name = "httpx", extra = ["socks"] }, @@ -1564,40 +1602,20 @@ dependencies = [ { name = "jsonschema" }, { name = "openai" }, { name = "pydantic" }, + { name = "pydantic-settings" }, { name = "python-dotenv" }, { name = "tiktoken" }, { name = "tokenizers" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/97/dd/28024c0e4cf2dc6ab1bad59b8357af7f460e952c69526eae28f12ac4ee5e/litellm-1.93.2.tar.gz", hash = "sha256:c5d5223ef07f36e0886397fb45cc9db4150f86a0c6f6835cee1d5524cab69dfd", size = 15955441, upload-time = "2026-08-09T02:17:49.646Z" } +sdist = { url = "https://files.pythonhosted.org/packages/26/c9/cb2730c6c763233e322fe7c5b2f53783eb10893cea9304e5474f1f20c306/litellm-1.101.2.tar.gz", hash = "sha256:790adf4ce19116d7bf4342492b1be5a90dd56e08d89795979bf6c1c3446a9670", size = 17493188, upload-time = "2026-09-24T00:04:22.712Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/64/c7/cb3f49dc60d57dda7fe368310fd5da2a94ec9b6a746bcf343a61e10bdeda/litellm-1.93.2-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:1bd0690efc94357e559de97927fd98437555cd5b5dd832544cfcca87297ccb80", size = 19938326, upload-time = "2026-08-09T02:16:38.041Z" }, - { url = "https://files.pythonhosted.org/packages/0c/bd/d77184fdaaf57d67d65da91dcfc61c7f656703e7ce4f950e07523e7de4e3/litellm-1.93.2-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:845ececc628737909b1422d1af18bd19ae453727a66244aa9da3ca37a3773111", size = 19862606, upload-time = "2026-08-09T02:16:40.653Z" }, - { url = "https://files.pythonhosted.org/packages/53/99/d8dd58b6840754a13cc2e1111b283aa28cbfc0ccc653a8725050916bb08e/litellm-1.93.2-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:498f9878ea773305e0638b6159d7e1ef27bb0b9a4292538d6634312d18a4e781", size = 20168532, upload-time = "2026-08-09T02:16:42.997Z" }, - { url = "https://files.pythonhosted.org/packages/d7/ca/559ca0f5e0b99b9f641086ae924c782f8d521d09384fbe9abbe0bddb6e61/litellm-1.93.2-cp311-cp311-manylinux_2_28_x86_64.whl", hash = "sha256:1e5618ef495b2e02299b376ca84ffb2647837aafee478cf3a1be17d47a8f0f73", size = 20162696, upload-time = "2026-08-09T02:16:45.283Z" }, - { url = "https://files.pythonhosted.org/packages/92/3e/18c31b27c7d1271b43bdc8ffbef01bfba68d90248bbe60bb2130dd17e43c/litellm-1.93.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:c2da463d70c9fffbea9532fd000e035328f5266b399a2fb4c6c76b3470478337", size = 20233518, upload-time = "2026-08-09T02:16:47.87Z" }, - { url = "https://files.pythonhosted.org/packages/d9/98/a6bae7c52f09cd03487a040f98eeedb899b3cf3fc541b87c6d051ee92e0d/litellm-1.93.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:2cf122399f84f8f04621ed6ef8f276dd6d61f4fab108932ce0e30368de34dd42", size = 20291180, upload-time = "2026-08-09T02:16:50.549Z" }, - { url = "https://files.pythonhosted.org/packages/77/2d/81d974f2533cf039afda7e3e0f769dc73dc692c75ec867cf29ec6f41c06f/litellm-1.93.2-cp311-cp311-win_amd64.whl", hash = "sha256:8eaaf780fab9a19234735ef94225172179d15bc28b67ddbec125194249a504b7", size = 19775654, upload-time = "2026-08-09T02:16:53.162Z" }, - { url = "https://files.pythonhosted.org/packages/d0/05/72fd8051f0f2f3c84b90986e6f4551db7c8b190ba3300f111461b7701689/litellm-1.93.2-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:3bf532c164ad7cb1b76f2c62afefdcc656b9b296374d075a4150e2ce10bb74c3", size = 19937403, upload-time = "2026-08-09T02:16:55.545Z" }, - { url = "https://files.pythonhosted.org/packages/9e/4d/5081b39bdb73cab04f8a86294a4534a029cf0434ac6932c7ae8049d55723/litellm-1.93.2-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:526b7afc037f79dfdd5c607f5085ac597c7fd301a6dedabea40baae899b27f19", size = 19853652, upload-time = "2026-08-09T02:16:57.977Z" }, - { url = "https://files.pythonhosted.org/packages/70/3f/fb70691266a7fd08c202406abea0153e82fa17f134cd9d58e4029cc741db/litellm-1.93.2-cp312-cp312-manylinux_2_28_aarch64.whl", hash = "sha256:294ad19f356f821ce97a5428d09439be5f38d22b218c73008d8a49e3e42eb145", size = 20165680, upload-time = "2026-08-09T02:17:00.65Z" }, - { url = "https://files.pythonhosted.org/packages/81/91/84424ce2a25595463e5d24e9cf8949877cd4ce93c0fcbf6486ecd685094f/litellm-1.93.2-cp312-cp312-manylinux_2_28_x86_64.whl", hash = "sha256:6f6a5e3907f0a1c9d8ff8d71a6cbac8a592e47a40da3f97167074947b5ba7d11", size = 20157772, upload-time = "2026-08-09T02:17:03.027Z" }, - { url = "https://files.pythonhosted.org/packages/8f/8d/b0eac7ee6d174564f820565c8c9a726ae83dbb8c4d3522daf175b95da002/litellm-1.93.2-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:8541f1b7fd5c437ad249ad68d0a11f68e5e2866b0649da5fa7d63b595e9b8b22", size = 20229256, upload-time = "2026-08-09T02:17:05.271Z" }, - { url = "https://files.pythonhosted.org/packages/ee/6d/03e931c1cb2d1e1b7a968de21aa9e4db853928200da856c35c940ee6faa9/litellm-1.93.2-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:712c9387419d7b06a10df59973f5e530592d61b2314102b0fa3142f3743f9a9e", size = 20287257, upload-time = "2026-08-09T02:17:08.175Z" }, - { url = "https://files.pythonhosted.org/packages/16/05/6c0fe2fcf31c260474c55fabe4ecb0e9e1343c9b9132e28589391b2ad33e/litellm-1.93.2-cp312-cp312-win_amd64.whl", hash = "sha256:cc0d58ccabd22ef7ef44a9e6f7247deb54ae42f5e126e6f00360c2b28b41bc2b", size = 19772580, upload-time = "2026-08-09T02:17:11.254Z" }, - { url = "https://files.pythonhosted.org/packages/70/74/e9046cffa69b32b710452480598e418b26a29896ece680c80ec23997fd16/litellm-1.93.2-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:f4071bef03e4c2942cd2ddc752727345b85447d6a7fee1ff5a4f8b92187966b0", size = 19938095, upload-time = "2026-08-09T02:17:13.929Z" }, - { url = "https://files.pythonhosted.org/packages/fa/db/6ef38a7a2f73d5cc507423954fa535a8546ead375c4c71265c093bdb4e9e/litellm-1.93.2-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:8a99ac7c0c1b78acd6bfd1959e9f203dca71fdbceb5f0c8691c2ad8eee450d7d", size = 19854187, upload-time = "2026-08-09T02:17:16.588Z" }, - { url = "https://files.pythonhosted.org/packages/cb/b3/80ee0143b88e2921f8c8f24c7331478258a8bf25a3d4d4450bd96043403e/litellm-1.93.2-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:a81ceff44c58ef504ab8bd787d03b82618765b9cfd530942386ae6d23c58be94", size = 20166307, upload-time = "2026-08-09T02:17:19.078Z" }, - { url = "https://files.pythonhosted.org/packages/98/60/cb326e1094f7042f28f9e21543d9f367a8aa25af6915bf4253b77da5c2a2/litellm-1.93.2-cp313-cp313-manylinux_2_28_x86_64.whl", hash = "sha256:dee1b02b7f52a5a408bf7c8d499f0834e49194651743758a511dcdd926c0b692", size = 20158336, upload-time = "2026-08-09T02:17:21.507Z" }, - { url = "https://files.pythonhosted.org/packages/b1/87/bad75146863531172c9dbae189486c7f4425b56a6641b55ab20745316048/litellm-1.93.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:d2edfa14b99bce706b35981703692e3ee631f9b87bf6dc28fb53b574f6480b20", size = 20229711, upload-time = "2026-08-09T02:17:24.073Z" }, - { url = "https://files.pythonhosted.org/packages/df/28/040b1853021ed8fd57be19eb2affb024d168951fe7e7abdbad91da3f6f3f/litellm-1.93.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:ae75a61c9abc827aa3131b7e640c952367a450830bb7c531b426b4ec2bb45f85", size = 20287584, upload-time = "2026-08-09T02:17:26.542Z" }, - { url = "https://files.pythonhosted.org/packages/d9/0b/4208815b0d666636cbf7afbd571eec3004d3a15d3150a23a9009fc2ce930/litellm-1.93.2-cp313-cp313-win_amd64.whl", hash = "sha256:c54a09ab20f94120a9d60a30d9970439dcefa00d2565d190505ff006a80c7a69", size = 19772641, upload-time = "2026-08-09T02:17:29.308Z" }, - { url = "https://files.pythonhosted.org/packages/09/4a/ff7a9c000519d2bab362318bf744a24c2500228e5fceaa6ac23acab96fa0/litellm-1.93.2-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:204cb0763fff9285bc87eb2dc0fc59b591999e5d94863d0964f806424d3c0cd6", size = 19943639, upload-time = "2026-08-09T02:17:31.811Z" }, - { url = "https://files.pythonhosted.org/packages/c4/26/29e9276ce4aa8ed133d9fd5ecc07375017d2228215547c6bbb17ccbc59b4/litellm-1.93.2-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:3126c84361606b9fb07fde7d57eccd8a1747304d64c4143e2e5e40ae6e7693fb", size = 19855435, upload-time = "2026-08-09T02:17:34.376Z" }, - { url = "https://files.pythonhosted.org/packages/f8/20/2c9c818248ae019b2d496ca41900a9a5651ab05e2400794cd8dc8b89b6d2/litellm-1.93.2-cp314-cp314-manylinux_2_28_aarch64.whl", hash = "sha256:1c84f7c4acb4e926a79b93145ab23231b300fc687bde7172ef884fc52d6011e0", size = 20166947, upload-time = "2026-08-09T02:17:36.828Z" }, - { url = "https://files.pythonhosted.org/packages/50/af/4016682be48350407837941ad1a1ae8185cca65b102e04e89eee2a2abccb/litellm-1.93.2-cp314-cp314-manylinux_2_28_x86_64.whl", hash = "sha256:cacf35cf703b12c54516fc6464a3e08c6dbb1dcfb97239e1f629294fe36a1cba", size = 20160055, upload-time = "2026-08-09T02:17:39.674Z" }, - { url = "https://files.pythonhosted.org/packages/5b/b5/c25d7fbe08490d8211bd6b69af23f3a922b68ad8c87c776480b0de64a505/litellm-1.93.2-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:0a7f3e5138e307e429bd8fa29cc0c48bb1e2b827792e8f7799ca4c8cff736103", size = 20230910, upload-time = "2026-08-09T02:17:42.159Z" }, - { url = "https://files.pythonhosted.org/packages/21/27/341b18a40d4d98a2ac09025c248a3a7edddaf15ce4096ac4a783ff2f70db/litellm-1.93.2-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:d8684629be3f7b5f8e2b6e5fe5ea27ff957c63a8d525d81c1460d8436e2e1857", size = 20288903, upload-time = "2026-08-09T02:17:44.433Z" }, - { url = "https://files.pythonhosted.org/packages/8d/45/dd9ef72075a83854f852b1bf9a97ec7029a2be9fb4e338fc6623eb09fc90/litellm-1.93.2-cp314-cp314-win_amd64.whl", hash = "sha256:a783b8b18ed68cb6a3b79d2b00273ec21aef92442e9b2712a50036cb84bfe583", size = 19772974, upload-time = "2026-08-09T02:17:46.972Z" }, + { url = "https://files.pythonhosted.org/packages/44/a7/4bccec0ac9cb1b2e94e391b666458d07480d342039c66383ac191819e8e7/litellm-1.101.2-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:48c42c2c2cf9d4b0d75f4e1670b1b64b9e0513488d0b737fe057fda0bc716551", size = 23827328, upload-time = "2026-09-24T00:04:03.196Z" }, + { url = "https://files.pythonhosted.org/packages/3d/3d/faf394e5ac5a1469de5cbd3939c9e330f744e648351934b981078ccc40d9/litellm-1.101.2-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:77195c8ed502c052bb31d4c3887356a308ac2e8c3b0b30c97e2b04b0be10dd44", size = 23484770, upload-time = "2026-09-24T00:04:06.248Z" }, + { url = "https://files.pythonhosted.org/packages/ba/84/60f70aa2683666626c4abe7aa44b52acca52cad911ee870ea244fd3b0796/litellm-1.101.2-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:abb7b3ac04f56ced46e53cca2369a5dd29539cab9fdcd6cc9a94987aa55a38d1", size = 23618000, upload-time = "2026-09-24T00:04:08.827Z" }, + { url = "https://files.pythonhosted.org/packages/e2/8e/c57a4e157f97b1bcef9b410d51e17507047bbb11c676f5c81b22e7190c7c/litellm-1.101.2-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:210c89194225778759aa6649f5c0d605572bf14708ec85712462ace479f47f04", size = 23994795, upload-time = "2026-09-24T00:04:12.292Z" }, + { url = "https://files.pythonhosted.org/packages/b3/49/8737aee5a5a15cac7eb8a972b800529923e2837bbbadf0617f56111344ab/litellm-1.101.2-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:144dd8d1ead7174a718d1748deffcda7438cf7dfe8dc7c20761b72c35117e9a3", size = 23693332, upload-time = "2026-09-24T00:04:14.886Z" }, + { url = "https://files.pythonhosted.org/packages/04/50/4e711caa0374309d6aaf5696549449c2078f0225dd36a22ee0ca44dd068f/litellm-1.101.2-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:ae95e7ef15e109472f2cec69da6028a874b416e56a704cd9b797362993fc63cf", size = 24092655, upload-time = "2026-09-24T00:04:17.677Z" }, + { url = "https://files.pythonhosted.org/packages/c1/7d/32d391ddcb30d4d5d08fddd0abe918e836f9b3f753237c2b12ecb3d7425a/litellm-1.101.2-cp310-abi3-win_amd64.whl", hash = "sha256:0f5ee6daf9082b7efca1dc851c10c0d4884a2f1e0d7ea410bd508961b5a2cdae", size = 23894930, upload-time = "2026-09-24T00:04:20.432Z" }, ] [[package]] @@ -2439,6 +2457,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/bc/16/4ea354101abb1287856baa4af2732be351c7bee728065aed451b678153fd/pytest_cov-6.2.1-py3-none-any.whl", hash = "sha256:f5bc4c23f42f1cdd23c70b1dab1bbaef4fc505ba950d53e0081d0730dd7e86d5", size = 24644, upload-time = "2025-06-12T10:47:45.932Z" }, ] +[[package]] +name = "python-dateutil" +version = "2.9.0.post0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "six" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/66/c0/0c8b6ad9f17a802ee498c46e004a0eb49bc148f2fd230864601a86dcf6db/python-dateutil-2.9.0.post0.tar.gz", hash = "sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3", size = 342432, upload-time = "2024-03-01T18:36:20.211Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ec/57/56b9bcc3c9c6a792fcbaf139543cee77261f3651ca9da0c93f5c1221264b/python_dateutil-2.9.0.post0-py2.py3-none-any.whl", hash = "sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427", size = 229892, upload-time = "2024-03-01T18:36:18.57Z" }, +] + [[package]] name = "python-dotenv" version = "1.2.2" @@ -2722,7 +2752,7 @@ requires-dist = [ { name = "greenlet", specifier = ">=3.2.1" }, { name = "h11", specifier = ">=0.16" }, { name = "httpx", extras = ["socks"], specifier = ">=0.28.1" }, - { name = "litellm", specifier = ">=1.93.0,<1.94" }, + { name = "litellm", specifier = ">=1.101.2,<1.102" }, { name = "marshmallow", specifier = ">=3.13,<4.0" }, { name = "mdurl", specifier = "==0.1.2" }, { name = "nostr-sdk", specifier = ">=0.45.1,<0.46" }, @@ -2882,6 +2912,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/4c/9b/0b8aa09817b63e78d94b4977f18b1fcaead3165a5ee49251c5d5c245bb2d/ruff-0.12.7-py3-none-win_arm64.whl", hash = "sha256:dfce05101dbd11833a0776716d5d1578641b7fddb537fe7fa956ab85d1769b69", size = 11982083, upload-time = "2025-07-29T22:32:33.881Z" }, ] +[[package]] +name = "s3transfer" +version = "0.19.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "botocore" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/76/43/35e4d8aa320bffe8287fe8f65f578fa2d2db0a64212f0e710dce58267854/s3transfer-0.19.2.tar.gz", hash = "sha256:ba0309fd86be3c27dbf78cdd813c13c5e1df16e5874b99d2535ebbdfb9892993", size = 165592, upload-time = "2026-07-22T19:30:44.432Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bc/e7/5c595c75e9f41a44f30e526eda465ea0b4eec93470e074e4a111b253f13a/s3transfer-0.19.2-py3-none-any.whl", hash = "sha256:d8168eccca828cbb2cd573675333f3bddd254313a9c42494b84c76b539e8ba25", size = 90216, upload-time = "2026-07-22T19:30:43.251Z" }, +] + [[package]] name = "setuptools" version = "84.0.0" From c365a30f3086710c0e4c59d2a31c7f6ad42acfc0 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 00:13:07 +0200 Subject: [PATCH 03/12] fix: make Claude Code work on Venice with prompt caching and gpt-6 --- routstr/upstream/base.py | 6 + routstr/upstream/messages_dispatch.py | 25 ++- routstr/upstream/venice.py | 110 ++++++++- tests/unit/test_venice_encrypted_reasoning.py | 210 ++++++++++++++++++ tests/unit/test_venice_system_cache.py | 86 +++++++ 5 files changed, 428 insertions(+), 9 deletions(-) create mode 100644 tests/unit/test_venice_encrypted_reasoning.py create mode 100644 tests/unit/test_venice_system_cache.py diff --git a/routstr/upstream/base.py b/routstr/upstream/base.py index 8f46d90b..2f65ebd3 100644 --- a/routstr/upstream/base.py +++ b/routstr/upstream/base.py @@ -2612,6 +2612,11 @@ class BaseUpstreamProvider: ) -> dict: return await messages_dispatch.aggregate_anthropic_events_to_message(iterator) + def transform_messages_stream( + self, stream: AsyncIterator[Any] + ) -> AsyncIterator[Any]: + return stream + def adapt_messages_request(self, body: dict, model_obj: Model) -> str: """Rewrite an allowlisted /v1/messages body for this upstream. @@ -2637,6 +2642,7 @@ class BaseUpstreamProvider: provider_prefix=self.get_litellm_provider_prefix(), transform_model_name=self.transform_model_name, adapt_request=lambda body: self.adapt_messages_request(body, model_obj), + transform_stream=self.transform_messages_stream, log_extra=log_extra, ) diff --git a/routstr/upstream/messages_dispatch.py b/routstr/upstream/messages_dispatch.py index 6de5e57f..d9df0277 100644 --- a/routstr/upstream/messages_dispatch.py +++ b/routstr/upstream/messages_dispatch.py @@ -406,16 +406,9 @@ def annotate_event(event: dict, requested_model: str | None) -> AnnotatedEvent: _coerce_float(root_cost_details.get("output_cost")), ) - event_type = str(event.get("type") or "") - payload = json.dumps(event) - if event_type: - sse_bytes = f"event: {event_type}\ndata: {payload}\n\n".encode() - else: - sse_bytes = f"data: {payload}\n\n".encode() - return AnnotatedEvent( event, - sse_bytes, + encode_sse(event), in_tokens, out_tokens, cache_read_tokens, @@ -427,6 +420,14 @@ def annotate_event(event: dict, requested_model: str | None) -> AnnotatedEvent: ) +def encode_sse(event: dict) -> bytes: + event_type = str(event.get("type") or "") + payload = json.dumps(event) + if event_type: + return f"event: {event_type}\ndata: {payload}\n\n".encode() + return f"data: {payload}\n\n".encode() + + async def stream_annotated_events( iterator: AsyncIterator[Any], requested_model: str | None, @@ -493,6 +494,7 @@ async def dispatch_anthropic_messages( provider_prefix: str, transform_model_name: Callable[[str], str], adapt_request: Callable[[dict], str] | None = None, + transform_stream: Callable[[AsyncIterator[Any]], AsyncIterator[Any]] | None = None, log_extra: dict[str, Any] | None = None, ) -> tuple[bool, Any, str | None]: """Call ``litellm.anthropic.messages.acreate`` and return @@ -505,6 +507,10 @@ async def dispatch_anthropic_messages( may rewrite it in place and returns a suffix for the upstream model name, which is how a provider expresses a feature litellm would otherwise translate into a parameter the upstream rejects. + + ``transform_stream`` rewrites the upstream event stream before it is + aggregated or handed to the client, so a provider can repair events + litellm translates faithfully but clients cannot use. """ if not request_body: raise UpstreamError("Missing request body for /v1/messages", status_code=400) @@ -640,6 +646,9 @@ async def dispatch_anthropic_messages( from_upstream_response=True, ) from exc + if transform_stream is not None and hasattr(result, "__aiter__"): + result = transform_stream(cast(AsyncIterator[Any], result)) + if not client_stream and hasattr(result, "__aiter__"): # Client asked for a non-streaming response but we always stream # from upstream — drain the events into a single Anthropic Message diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index 379a4a66..15a476fa 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -1,13 +1,16 @@ from __future__ import annotations -from typing import TYPE_CHECKING, Any +from collections.abc import AsyncGenerator, AsyncIterator +from typing import TYPE_CHECKING, Any, cast import httpx from ..core.exceptions import UpstreamError from ..core.logging import get_logger from ..payment.models import Architecture, Model, Pricing, TopProvider +from . import messages_dispatch from .base import BaseUpstreamProvider +from .stream_ownership import aclose_if_needed if TYPE_CHECKING: from ..core.db import UpstreamProviderRow @@ -50,6 +53,73 @@ _UNENFORCEABLE_WEB_SEARCH_KEYS = frozenset( {"allowed_domains", "blocked_domains", "user_location"} ) +# Venice streams OpenAI reasoning models' encrypted reasoning as a trailing +# ``reasoning_content`` delta carrying this marker. litellm turns it into a +# plaintext ``thinking`` block after the answer, which clients render as +# gibberish and which makes Claude Code report an empty final result. +_ENCRYPTED_REASONING_MARKER = "__ENCRYPTED_REASONING__" + + +async def _drop_encrypted_reasoning( + upstream: AsyncIterator[Any], +) -> AsyncGenerator[bytes, None]: + """A thinking block's start carries no text, so it is held until its first + delta shows whether it is the encrypted payload; later indices shift down + to close the gap.""" + encode = messages_dispatch.encode_sse + sse_buffer = b"" + dropped: set[int] = set() + held: list[dict] | None = None + held_index: int | None = None + + def shift(event: dict) -> dict: + index = event.get("index") + if not isinstance(index, int): + return event + gap = sum(1 for d in dropped if d < index) + return {**event, "index": index - gap} if gap else event + + try: + async for chunk in upstream: + events, sse_buffer = messages_dispatch.events_from_chunk(chunk, sse_buffer) + for event in events: + etype = event.get("type") + index = event.get("index") + if held is not None: + delta = event.get("delta") or {} + is_own_delta = ( + index == held_index and etype == "content_block_delta" + ) + thinking = str(delta.get("thinking") or "") + if is_own_delta and thinking.startswith( + _ENCRYPTED_REASONING_MARKER + ): + dropped.add(cast(int, index)) + held = None + continue + if is_own_delta and not thinking: + held.append(event) + continue + for pending in held: + yield encode(shift(pending)) + held = None + if index in dropped: + continue + block = event.get("content_block") or {} + if ( + etype == "content_block_start" + and block.get("type") == "thinking" + and not block.get("thinking") + ): + held, held_index = [event], index + continue + yield encode(shift(event)) + if held is not None: + for pending in held: + yield encode(shift(pending)) + finally: + await aclose_if_needed(upstream) + def _is_web_search_tool(tool: Any) -> bool: """An Anthropic server-side web-search tool, by either of its markers. @@ -66,6 +136,35 @@ def _is_web_search_tool(tool: Any) -> bool: ) or tool.get("name") == "web_search" +def _merge_cache_marked_system(body: dict) -> None: + """Venice rejects an OpenAI ``system`` message with two or more text parts + when any part carries ``cache_control`` (``400 system: text content blocks + must contain non-whitespace text``), even though every part is non-blank. + Claude Code always sends that shape. A single marked block is accepted and + still caches, so the prefix stays cacheable under the last marker. + """ + system = body.get("system") + if not isinstance(system, list) or len(system) < 2: + return + if not all( + isinstance(block, dict) + and block.get("type") == "text" + and isinstance(block.get("text"), str) + for block in system + ): + return + markers = [block["cache_control"] for block in system if block.get("cache_control")] + if not markers: + return + body["system"] = [ + { + "type": "text", + "text": "\n\n".join(block["text"] for block in system), + "cache_control": markers[-1], + } + ] + + def _usd(entry: Any) -> float | None: """Read the USD leg of a Venice ``{usd, diem}`` price pair.""" if isinstance(entry, dict): @@ -114,7 +213,16 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): def transform_model_name(self, model_id: str) -> str: return model_id.removeprefix("venice/") + def transform_messages_stream( + self, stream: AsyncIterator[Any] + ) -> AsyncIterator[Any]: + return _drop_encrypted_reasoning(stream) + def adapt_messages_request(self, body: dict, model_obj: Model) -> str: + _merge_cache_marked_system(body) + return self._adapt_web_search(body) + + def _adapt_web_search(self, body: dict) -> str: """Trade an Anthropic web-search tool for Venice's own search switch. Left in the body, litellm's Anthropic adapter rewrites the tool into a diff --git a/tests/unit/test_venice_encrypted_reasoning.py b/tests/unit/test_venice_encrypted_reasoning.py new file mode 100644 index 00000000..43af7614 --- /dev/null +++ b/tests/unit/test_venice_encrypted_reasoning.py @@ -0,0 +1,210 @@ +import json +from collections.abc import AsyncIterator +from typing import Any +from unittest.mock import AsyncMock, patch + +import pytest + +from routstr.upstream import messages_dispatch +from routstr.upstream.base import BaseUpstreamProvider +from routstr.upstream.venice import VeniceUpstreamProvider, _drop_encrypted_reasoning + +from .test_venice_web_search import _model + +ENCRYPTED = "__ENCRYPTED_REASONING__id=rs_0b04\ngAAAAABqvDJD" + + +def _block(index: int, block: dict, deltas: list[dict]) -> list[dict]: + return [ + {"type": "content_block_start", "index": index, "content_block": block}, + *({"type": "content_block_delta", "index": index, "delta": d} for d in deltas), + {"type": "content_block_stop", "index": index}, + ] + + +def _thinking(index: int, text: str) -> list[dict]: + return _block( + index, + {"type": "thinking", "thinking": "", "signature": ""}, + [{"type": "thinking_delta", "thinking": text}], + ) + + +def _text(index: int, text: str) -> list[dict]: + return _block( + index, + {"type": "text", "text": ""}, + [{"type": "text_delta", "text": text}], + ) + + +def _tool(index: int) -> list[dict]: + return _block( + index, + {"type": "tool_use", "id": "call_1", "name": "Bash", "input": {}}, + [{"type": "input_json_delta", "partial_json": '{"command":"ls"}'}], + ) + + +def _message(blocks: list[dict], stop_reason: str = "end_turn") -> list[dict]: + return [ + { + "type": "message_start", + "message": {"id": "msg_1", "role": "assistant", "content": []}, + }, + *blocks, + {"type": "message_delta", "delta": {"stop_reason": stop_reason}}, + {"type": "message_stop"}, + ] + + +async def _upstream(events: list[dict], *, split: bool = False) -> AsyncIterator[Any]: + payload = b"".join(messages_dispatch.encode_sse(e) for e in events) + if split: + for i in range(0, len(payload), 7): + yield payload[i : i + 7] + else: + yield payload + + +async def _filtered(events: list[dict], **kwargs: Any) -> list[dict]: + buffer = b"" + out: list[dict] = [] + async for chunk in _drop_encrypted_reasoning(_upstream(events, **kwargs)): + parsed, buffer = messages_dispatch.events_from_chunk(chunk, buffer) + out.extend(parsed) + return out + + +def _starts(events: list[dict]) -> list[tuple[int, str]]: + return [ + (e["index"], e["content_block"]["type"]) + for e in events + if e["type"] == "content_block_start" + ] + + +@pytest.mark.asyncio +async def test_trailing_encrypted_reasoning_is_dropped() -> None: + events = _message([*_text(0, "a.txt contains: hello"), *_thinking(1, ENCRYPTED)]) + + out = await _filtered(events) + + assert _starts(out) == [(0, "text")] + assert all(ENCRYPTED not in json.dumps(e) for e in out) + assert out[-2]["delta"]["stop_reason"] == "end_turn" + + +@pytest.mark.asyncio +async def test_leading_encrypted_reasoning_closes_index_gap() -> None: + events = _message( + [*_thinking(0, ENCRYPTED), *_text(1, "hi"), *_tool(2)], "tool_use" + ) + + out = await _filtered(events, split=True) + + assert _starts(out) == [(0, "text"), (1, "tool_use")] + assert {e["index"] for e in out if "index" in e} == {0, 1} + + +@pytest.mark.asyncio +async def test_plaintext_thinking_is_kept_in_order() -> None: + events = _message([*_thinking(0, "Let me list files."), *_tool(1)], "tool_use") + + out = await _filtered(events) + + assert out == events + + +@pytest.mark.asyncio +async def test_thinking_start_without_delta_is_flushed() -> None: + events = _message( + [ + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "thinking", "thinking": "", "signature": ""}, + }, + {"type": "content_block_stop", "index": 0}, + *_text(1, "ok"), + ] + ) + + out = await _filtered(events) + + assert out == events + + +@pytest.mark.asyncio +async def test_aggregated_message_ends_with_answer_text() -> None: + events = _message([*_text(0, "hello"), *_thinking(1, ENCRYPTED)]) + + message = await messages_dispatch.aggregate_anthropic_events_to_message( + _drop_encrypted_reasoning(_upstream(events)) + ) + + assert [b["type"] for b in message["content"]] == ["text"] + assert message["content"][0]["text"] == "hello" + + +async def _dispatched_blocks( + provider: BaseUpstreamProvider, *, stream: bool +) -> list[str]: + events = _message([*_text(0, "hello"), *_thinking(1, ENCRYPTED)]) + with patch( + "litellm.anthropic.messages.acreate", + new=AsyncMock(return_value=_upstream(events)), + ): + _, result, _ = await provider._dispatch_anthropic_messages( + request_body=json.dumps( + { + "model": "x", + "stream": stream, + "max_tokens": 64, + "messages": [{"role": "user", "content": "hi"}], + } + ).encode(), + model_obj=_model(), + ) + if not stream: + return [b["type"] for b in result["content"]] + buffer = b"" + out: list[dict] = [] + async for chunk in result: + parsed, buffer = messages_dispatch.events_from_chunk(chunk, buffer) + out.extend(parsed) + return [t for _, t in _starts(out)] + + +@pytest.mark.asyncio +@pytest.mark.parametrize("stream", [True, False]) +async def test_venice_dispatch_drops_encrypted_reasoning(stream: bool) -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + + assert await _dispatched_blocks(provider, stream=stream) == ["text"] + + +@pytest.mark.asyncio +async def test_other_providers_keep_thinking_blocks() -> None: + provider = BaseUpstreamProvider(base_url="https://example.com/v1", api_key="k") + + assert await _dispatched_blocks(provider, stream=True) == ["text", "thinking"] + + +@pytest.mark.asyncio +async def test_closing_the_filter_closes_upstream() -> None: + closed = False + + async def upstream() -> AsyncIterator[bytes]: + nonlocal closed + try: + for event in _message(_text(0, "hello")): + yield messages_dispatch.encode_sse(event) + finally: + closed = True + + filtered = _drop_encrypted_reasoning(upstream()) + await filtered.__anext__() + await filtered.aclose() + + assert closed diff --git a/tests/unit/test_venice_system_cache.py b/tests/unit/test_venice_system_cache.py new file mode 100644 index 00000000..6b8cc67f --- /dev/null +++ b/tests/unit/test_venice_system_cache.py @@ -0,0 +1,86 @@ +from __future__ import annotations + +import pytest + +from routstr.upstream.venice import VeniceUpstreamProvider + +from .test_venice_web_search import _body, _dispatch + +EPHEMERAL = {"type": "ephemeral"} + +CLAUDE_CODE_SYSTEM = [ + { + "type": "text", + "text": "x-anthropic-billing-header: cc_version=2.1.281; cc_entrypoint=cli;", + }, + {"type": "text", "text": "You are a Claude agent.", "cache_control": EPHEMERAL}, + { + "type": "text", + "text": "\nYou are an interactive agent.", + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + }, +] + + +@pytest.mark.asyncio +async def test_cache_marked_multi_block_system_is_merged_into_one_block() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + + kwargs = await _dispatch(provider, _body(system=CLAUDE_CODE_SYSTEM)) + + assert kwargs["system"] == [ + { + "type": "text", + "text": ( + "x-anthropic-billing-header: cc_version=2.1.281; cc_entrypoint=cli;" + "\n\nYou are a Claude agent.\n\n\nYou are an interactive agent." + ), + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + } + ] + + +@pytest.mark.asyncio +async def test_unmarked_multi_block_system_is_untouched() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + system = [{"type": "text", "text": "A."}, {"type": "text", "text": "B."}] + + kwargs = await _dispatch(provider, _body(system=system)) + + assert kwargs["system"] == system + + +@pytest.mark.asyncio +async def test_single_marked_block_and_string_system_are_untouched() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + single = [{"type": "text", "text": "A.", "cache_control": EPHEMERAL}] + + assert (await _dispatch(provider, _body(system=single)))["system"] == single + assert (await _dispatch(provider, _body(system="A.")))["system"] == "A." + + +@pytest.mark.asyncio +async def test_message_and_tool_cache_markers_are_kept() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + messages = [ + { + "role": "user", + "content": [{"type": "text", "text": "hi", "cache_control": EPHEMERAL}], + } + ] + tools = [ + { + "name": "Bash", + "description": "Run a command", + "input_schema": {"type": "object", "properties": {}}, + "cache_control": EPHEMERAL, + } + ] + + kwargs = await _dispatch( + provider, + _body(system=CLAUDE_CODE_SYSTEM, messages=messages, tools=tools), + ) + + assert kwargs["messages"] == messages + assert kwargs["tools"] == tools From a713f819616c2deab7bd8fb9d5772c38e635858d Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 02:15:57 +0200 Subject: [PATCH 04/12] perf: cache provider catalogs and load admin models page progressively --- routstr/core/admin.py | 92 ++++++++++++--- tests/conftest.py | 15 +++ tests/unit/test_admin_remote_models_cache.py | 112 +++++++++++++++++++ ui/app/model/loading.tsx | 23 ++++ ui/components/model-provider-section.tsx | 19 +++- ui/components/model-selector.tsx | 46 +++++--- ui/components/models-page.tsx | 30 +++-- ui/lib/api/services/admin.ts | 91 +++++++++------ ui/lib/hooks/use-models-with-providers.ts | 50 +++++++++ ui/lib/hooks/use-progressive-list.ts | 46 ++++++++ 10 files changed, 446 insertions(+), 78 deletions(-) create mode 100644 tests/unit/test_admin_remote_models_cache.py create mode 100644 ui/app/model/loading.tsx create mode 100644 ui/lib/hooks/use-models-with-providers.ts create mode 100644 ui/lib/hooks/use-progressive-list.ts diff --git a/routstr/core/admin.py b/routstr/core/admin.py index 365e2f68..f3862edc 100644 --- a/routstr/core/admin.py +++ b/routstr/core/admin.py @@ -1,6 +1,8 @@ +import asyncio import json import re import secrets +import time from datetime import datetime, timezone from pathlib import Path @@ -54,6 +56,9 @@ async def _refresh_provider_model_paths(upstream_provider_id: int) -> None: """Queue discovery sync without blocking the committed admin mutation.""" from ..upstream.model_paths import schedule_model_paths_refresh_for_provider + # Every provider/model mutation funnels through here, so it is also the one + # place that can keep the cached admin catalog from serving a stale listing. + invalidate_remote_models_cache(upstream_provider_id) await schedule_model_paths_refresh_for_provider(upstream_provider_id) @@ -1183,6 +1188,7 @@ async def delete_upstream_provider(provider_id: str) -> dict[str, object]: await session.delete(provider) await session.commit() + invalidate_remote_models_cache(deleted_id) await reinitialize_upstreams() await refresh_model_maps() return {"ok": True, "deleted_id": deleted_id} @@ -1196,13 +1202,78 @@ async def get_provider_types() -> list[dict[str, object]]: return [cls.get_provider_metadata() for cls in upstream_provider_classes] +# The admin catalog view is opened repeatedly and by several panels at once, +# while every miss costs a live upstream round trip. Keep the raw listing for a +# short window and let concurrent readers share one in-flight fetch. +_REMOTE_MODELS_TTL_SECONDS = 120.0 +_REMOTE_MODELS_FETCH_TIMEOUT_SECONDS = 20.0 +_remote_models_cache: dict[int, tuple[float, list]] = {} +_remote_models_locks: dict[int, asyncio.Lock] = {} +# Bumped on every invalidation so a fetch that started against the old provider +# config cannot write its result back after the cache was cleared. +_remote_models_generation = 0 + + +def invalidate_remote_models_cache(provider_pk: int | None = None) -> None: + global _remote_models_generation + _remote_models_generation += 1 + if provider_pk is None: + _remote_models_cache.clear() + _remote_models_locks.clear() + else: + _remote_models_cache.pop(provider_pk, None) + + +async def _get_remote_models( + provider: UpstreamProviderRow, provider_pk: int, force_refresh: bool = False +) -> list: + from ..upstream.helpers import _instantiate_provider + + now = time.monotonic() + cached = _remote_models_cache.get(provider_pk) + if not force_refresh and cached and now - cached[0] < _REMOTE_MODELS_TTL_SECONDS: + return cached[1] + + lock = _remote_models_locks.setdefault(provider_pk, asyncio.Lock()) + async with lock: + cached = _remote_models_cache.get(provider_pk) + now = time.monotonic() + if ( + not force_refresh + and cached + and now - cached[0] < _REMOTE_MODELS_TTL_SECONDS + ): + return cached[1] + + upstream_instance = _instantiate_provider(provider) + if not upstream_instance: + return [] + + generation = _remote_models_generation + try: + models = await asyncio.wait_for( + upstream_instance.fetch_models(), + timeout=_REMOTE_MODELS_FETCH_TIMEOUT_SECONDS, + ) + except Exception as e: + logger.error(f"Failed to fetch models from {provider.provider_type}: {e}") + # A stale listing beats an empty one for an operator view. + return cached[1] if cached else [] + + if generation == _remote_models_generation: + _remote_models_cache[provider_pk] = (time.monotonic(), models) + return models + + @admin_router.get( "/api/upstream-providers/{provider_id}/models", dependencies=[Depends(require_admin_api)], ) -async def get_provider_models(provider_id: str) -> dict[str, object]: - from ..upstream.helpers import _instantiate_provider - +async def get_provider_models( + provider_id: str, + include_remote: bool = Query(True), + refresh_remote: bool = Query(False), +) -> dict[str, object]: async with create_session() as session: provider = await _get_upstream_provider_by_ref(session, provider_id) provider_pk = _provider_pk(provider) @@ -1214,16 +1285,11 @@ async def get_provider_models(provider_id: str) -> dict[str, object]: apply_fees=False, ) - upstream_models = [] - upstream_instance = _instantiate_provider(provider) - if upstream_instance: - try: - raw_models = await upstream_instance.fetch_models() - upstream_models = raw_models - except Exception as e: - logger.error( - f"Failed to fetch models from {provider.provider_type}: {e}" - ) + upstream_models: list = [] + if include_remote: + upstream_models = await _get_remote_models( + provider, provider_pk, force_refresh=refresh_remote + ) db_model_ids = {model.id for model in db_models} filtered_remote_models = [ diff --git a/tests/conftest.py b/tests/conftest.py index d1bfa919..74c3d058 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -31,3 +31,18 @@ def _isolate_redemption_negative_cache() -> Iterator[None]: redemption_negative_cache.clear() yield redemption_negative_cache.clear() + + +@pytest.fixture(autouse=True) +def _isolate_admin_remote_models_cache() -> Iterator[None]: + """Clear the admin catalog cache between tests. + + Provider primary keys restart at 1 for every fresh test database, so a + cached listing from an earlier test would otherwise answer for a different + provider that happens to reuse the same key. + """ + from routstr.core.admin import invalidate_remote_models_cache + + invalidate_remote_models_cache() + yield + invalidate_remote_models_cache() diff --git a/tests/unit/test_admin_remote_models_cache.py b/tests/unit/test_admin_remote_models_cache.py new file mode 100644 index 00000000..6a22ce00 --- /dev/null +++ b/tests/unit/test_admin_remote_models_cache.py @@ -0,0 +1,112 @@ +"""Cache behavior of the admin provider catalog listing.""" + +import asyncio +from types import SimpleNamespace +from typing import Any +from unittest.mock import AsyncMock, patch + +import pytest + +from routstr.core import admin +from routstr.core.admin import _get_remote_models, invalidate_remote_models_cache + +PROVIDER = SimpleNamespace(provider_type="generic") + + +def _upstream(fetch: Any) -> Any: + return patch( + "routstr.upstream.helpers._instantiate_provider", + return_value=SimpleNamespace(fetch_models=fetch), + ) + + +@pytest.mark.asyncio +async def test_second_read_is_served_from_cache() -> None: + fetch = AsyncMock(return_value=["a"]) + with _upstream(fetch): + assert await _get_remote_models(PROVIDER, 1) == ["a"] # type: ignore[arg-type] + assert await _get_remote_models(PROVIDER, 1) == ["a"] # type: ignore[arg-type] + assert fetch.await_count == 1 + + +@pytest.mark.asyncio +async def test_force_refresh_and_invalidation_refetch() -> None: + fetch = AsyncMock(side_effect=[["a"], ["b"], ["c"]]) + with _upstream(fetch): + await _get_remote_models(PROVIDER, 1) # type: ignore[arg-type] + assert await _get_remote_models(PROVIDER, 1, force_refresh=True) == ["b"] # type: ignore[arg-type] + invalidate_remote_models_cache(1) + assert await _get_remote_models(PROVIDER, 1) == ["c"] # type: ignore[arg-type] + + +@pytest.mark.asyncio +async def test_expired_entry_is_refetched() -> None: + fetch = AsyncMock(side_effect=[["a"], ["b"]]) + with _upstream(fetch): + await _get_remote_models(PROVIDER, 1) # type: ignore[arg-type] + stamp, models = admin._remote_models_cache[1] + admin._remote_models_cache[1] = ( + stamp - admin._REMOTE_MODELS_TTL_SECONDS - 1, + models, + ) + assert await _get_remote_models(PROVIDER, 1) == ["b"] # type: ignore[arg-type] + + +@pytest.mark.asyncio +async def test_failed_refresh_falls_back_to_stale_listing() -> None: + fetch = AsyncMock(side_effect=[["a"], RuntimeError("upstream down")]) + with _upstream(fetch): + await _get_remote_models(PROVIDER, 1) # type: ignore[arg-type] + assert await _get_remote_models(PROVIDER, 1, force_refresh=True) == ["a"] # type: ignore[arg-type] + + +@pytest.mark.asyncio +async def test_failed_first_fetch_returns_empty_and_caches_nothing() -> None: + fetch = AsyncMock(side_effect=RuntimeError("upstream down")) + with _upstream(fetch): + assert await _get_remote_models(PROVIDER, 1) == [] # type: ignore[arg-type] + assert 1 not in admin._remote_models_cache + + +@pytest.mark.asyncio +async def test_concurrent_readers_share_one_fetch() -> None: + release = asyncio.Event() + + async def slow_fetch() -> list[str]: + await release.wait() + return ["a"] + + fetch = AsyncMock(side_effect=slow_fetch) + with _upstream(fetch): + readers = [ + asyncio.create_task(_get_remote_models(PROVIDER, 1)) # type: ignore[arg-type] + for _ in range(5) + ] + await asyncio.sleep(0) + release.set() + results = await asyncio.gather(*readers) + assert results == [["a"]] * 5 + assert fetch.await_count == 1 + + +@pytest.mark.asyncio +async def test_invalidation_during_fetch_discards_in_flight_result() -> None: + release = asyncio.Event() + + async def slow_fetch() -> list[str]: + await release.wait() + return ["old"] + + with _upstream(AsyncMock(side_effect=slow_fetch)): + reader = asyncio.create_task(_get_remote_models(PROVIDER, 1)) # type: ignore[arg-type] + await asyncio.sleep(0) + invalidate_remote_models_cache(1) + release.set() + assert await reader == ["old"] + assert 1 not in admin._remote_models_cache + + +@pytest.mark.asyncio +async def test_uninstantiable_provider_returns_empty() -> None: + with patch("routstr.upstream.helpers._instantiate_provider", return_value=None): + assert await _get_remote_models(PROVIDER, 1) == [] # type: ignore[arg-type] diff --git a/ui/app/model/loading.tsx b/ui/app/model/loading.tsx new file mode 100644 index 00000000..390be4d8 --- /dev/null +++ b/ui/app/model/loading.tsx @@ -0,0 +1,23 @@ +import { AppPageShell } from '@/components/app-page-shell'; +import { PageHeader } from '@/components/page-header'; +import { Skeleton } from '@/components/ui/skeleton'; + +/** + * Route-level fallback so clicking "Models" lands on the page immediately + * instead of holding the previous route until this one's chunk is parsed. + */ +export default function ModelPageLoading() { + return ( + +
+ + + + +
+
+ ); +} diff --git a/ui/components/model-provider-section.tsx b/ui/components/model-provider-section.tsx index 9f441678..3acd132e 100644 --- a/ui/components/model-provider-section.tsx +++ b/ui/components/model-provider-section.tsx @@ -1,5 +1,6 @@ import { useMemo } from 'react'; import type { Model } from '@/lib/api/schemas/models'; +import { useProgressiveList } from '@/lib/hooks/use-progressive-list'; import type { AdminModelGroup } from '@/lib/api/services/admin'; import type { DisplayUnit } from '@/lib/types/units'; import { ModelItemCard } from '@/components/model-item-card'; @@ -24,6 +25,7 @@ import { Edit3, Globe, Key, + Loader2, MoreVertical, RefreshCw, } from 'lucide-react'; @@ -103,10 +105,21 @@ export function ModelProviderSection({ }); }, [provider, providerModels]); + const { visibleItems: visibleProviderModels, hiddenCount } = + useProgressiveList(keyedProviderModels); + + const pendingRowsNotice = + hiddenCount > 0 ? ( +
+ + Rendering {hiddenCount} more model{hiddenCount === 1 ? '' : 's'}… +
+ ) : null; + if (filterProvider) { return (
- {keyedProviderModels.map(({ model, renderKey }) => ( + {visibleProviderModels.map(({ model, renderKey }) => ( onDeleteModel(model.id)} /> ))} + {pendingRowsNotice}
); } @@ -217,7 +231,7 @@ export function ModelProviderSection({
- {keyedProviderModels.map(({ model, renderKey }) => ( + {visibleProviderModels.map(({ model, renderKey }) => ( onDeleteModel(model.id)} /> ))} + {pendingRowsNotice}
diff --git a/ui/components/model-selector.tsx b/ui/components/model-selector.tsx index 8636fb36..32552ef3 100644 --- a/ui/components/model-selector.tsx +++ b/ui/components/model-selector.tsx @@ -1,7 +1,7 @@ 'use client'; import React, { useState, useMemo } from 'react'; -import { useQuery, useMutation, useQueryClient } from '@tanstack/react-query'; +import { useMutation, useQueryClient } from '@tanstack/react-query'; import { type Model, type GroupSettings } from '@/lib/api/schemas/models'; import { AdminService, @@ -13,6 +13,7 @@ import { AddProviderModelDialog } from '@/components/add-provider-model-dialog'; import { EditGroupForm } from '@/components/edit-group-form'; import { ModelProviderSection } from '@/components/model-provider-section'; import { useDisplayCurrency } from '@/lib/hooks/use-display-currency'; +import { useModelsWithProviders } from '@/lib/hooks/use-models-with-providers'; import { Button } from '@/components/ui/button'; import { Checkbox } from '@/components/ui/checkbox'; import { Skeleton } from '@/components/ui/skeleton'; @@ -27,7 +28,7 @@ import { AlertDialogHeader, AlertDialogTitle, } from '@/components/ui/alert-dialog'; -import { Trash2, Ban, CheckCircle, Plus } from 'lucide-react'; +import { Trash2, Ban, CheckCircle, Loader2, Plus } from 'lucide-react'; import { toast } from 'sonner'; import { sortModels, @@ -131,19 +132,15 @@ export function ModelSelector({ const queryClient = useQueryClient(); - // Fetch models and groups + // Shared with the page shell, so mounting this panel costs no extra fetch. const { - data: modelsData, + models, + groups, isLoading: isLoadingModels, + isFetchingRemote, error: modelsError, refetch: refetchModels, - } = useQuery({ - queryKey: ['models-with-providers'], - queryFn: () => AdminService.getModelsWithProviders(), - refetchOnWindowFocus: false, - }); - - const { models = [], groups = [] } = modelsData || {}; + } = useModelsWithProviders(); const allOverrideModels = useMemo( () => models.filter(isOverrideModel), [models] @@ -868,11 +865,19 @@ export function ModelSelector({ {Object.keys(groupedModels).length === 0 ? ( -
-

- Try broadening your search or switch to a different provider scope. -

-
+ isFetchingRemote ? ( +
+ + +
+ ) : ( +
+

+ Try broadening your search or switch to a different provider + scope. +

+
+ ) ) : null} {/* Provider Groups or Filtered Models */} @@ -922,6 +927,15 @@ export function ModelSelector({ ); })} + {/* The stored rows render first; provider catalogs arrive after their + upstream calls return, so the list says more is still on the way. */} + {isFetchingRemote && Object.keys(groupedModels).length > 0 ? ( +
+ + Loading provider catalogs… +
+ ) : null} + {/* Forms and Dialogs */} {modelDialogState.providerId && ( import('@/components/model-tester').then((m) => m.ModelTester), + { loading: () => , ssr: false } +); + +const ApiEndpointTester = dynamic( + () => + import('@/components/api-endpoint-tester').then((m) => m.ApiEndpointTester), + { loading: () => , ssr: false } +); + export function ModelsPage() { const [filteredModels, setFilteredModels] = useState( undefined @@ -31,16 +42,11 @@ export function ModelsPage() { useState('all'); const { - data: modelsData, + models, + groups, isLoading: isLoadingModels, error: modelsError, - } = useQuery({ - queryKey: ['admin-models-with-providers'], - queryFn: () => AdminService.getModelsWithProviders(), - refetchOnWindowFocus: false, - }); - - const { models = [], groups = [] } = modelsData || {}; + } = useModelsWithProviders(); const groupedModels = useMemo( () => groupAndSortModelsByProvider(models), diff --git a/ui/lib/api/services/admin.ts b/ui/lib/api/services/admin.ts index e510a571..78c8f01c 100644 --- a/ui/lib/api/services/admin.ts +++ b/ui/lib/api/services/admin.ts @@ -317,9 +317,14 @@ export class AdminService { ); } - static async getProviderModels(providerId: number): Promise { + static async getProviderModels( + providerId: number, + options: { includeRemote?: boolean } = {} + ): Promise { + const query = + options.includeRemote === false ? '?include_remote=false' : ''; const data = await apiClient.get( - `/admin/api/upstream-providers/${providerId}/models` + `/admin/api/upstream-providers/${providerId}/models${query}` ); // Convert pricing for all models in the list so the UI receives "per 1M tokens" values @@ -428,7 +433,9 @@ export class AdminService { ); } - static async getModelsWithProviders(): Promise<{ + static async getModelsWithProviders( + options: { includeRemote?: boolean } = {} + ): Promise<{ models: AdminModelAsModel[]; groups: AdminModelGroup[]; }> { @@ -446,14 +453,50 @@ export class AdminService { const allModels: AdminModelAsModel[] = []; const seenModelIds = new Set(); - for (const provider of providers) { - try { - const providerModels = await this.getProviderModels(provider.id); + // One provider's catalog never depends on another's, and each miss costs an + // upstream round trip, so the whole fan-out happens in a single wave. + const providerResults = await Promise.all( + providers.map(async (provider) => { + try { + return { + provider, + models: await this.getProviderModels(provider.id, options), + }; + } catch (error) { + console.error( + `Failed to fetch models for provider ${provider.id}:`, + error + ); + return null; + } + }) + ); - providerModels.db_models.forEach((dbModel) => { - seenModelIds.add(dbModel.id); + for (const result of providerResults) { + if (!result) { + continue; + } + const { provider, models: providerModels } = result; + providerModels.db_models.forEach((dbModel) => { + seenModelIds.add(dbModel.id); + const modelWithProvider = { + ...dbModel, + upstream_provider_id: provider.id, + }; + allModels.push({ + ...this.transformAdminModelToModel( + modelWithProvider, + provider.provider_type + ), + has_own_api_key: false, + api_key_type: 'group', + }); + }); + + providerModels.remote_models.forEach((remoteModel) => { + if (!seenModelIds.has(remoteModel.id)) { const modelWithProvider = { - ...dbModel, + ...remoteModel, upstream_provider_id: provider.id, }; allModels.push({ @@ -462,33 +505,11 @@ export class AdminService { provider.provider_type ), has_own_api_key: false, - api_key_type: 'group', + api_key_type: 'remote', + soft_deleted: false, }); - }); - - providerModels.remote_models.forEach((remoteModel) => { - if (!seenModelIds.has(remoteModel.id)) { - const modelWithProvider = { - ...remoteModel, - upstream_provider_id: provider.id, - }; - allModels.push({ - ...this.transformAdminModelToModel( - modelWithProvider, - provider.provider_type - ), - has_own_api_key: false, - api_key_type: 'remote', - soft_deleted: false, - }); - } - }); - } catch (error) { - console.error( - `Failed to fetch models for provider ${provider.id}:`, - error - ); - } + } + }); } return { models: allModels, groups }; diff --git a/ui/lib/hooks/use-models-with-providers.ts b/ui/lib/hooks/use-models-with-providers.ts new file mode 100644 index 00000000..90c2f1f5 --- /dev/null +++ b/ui/lib/hooks/use-models-with-providers.ts @@ -0,0 +1,50 @@ +'use client'; + +import { useQuery, useQueryClient } from '@tanstack/react-query'; +import { AdminService } from '@/lib/api/services/admin'; + +export const modelsWithProvidersQueryKey = ['models-with-providers'] as const; +const localModelsQueryKey = ['models-with-providers', 'local'] as const; + +/** + * Shared catalog read for every models view. + * + * The database rows come back without touching an upstream, so they render + * first; the listing that needs live provider calls replaces them once it + * lands. Both queries live under one key prefix, so a single + * `invalidateQueries(['models-with-providers'])` still refreshes the pair, and + * every consumer of this hook shares one request instead of fanning out again. + */ +export function useModelsWithProviders() { + const queryClient = useQueryClient(); + + const localQuery = useQuery({ + queryKey: localModelsQueryKey, + queryFn: () => + AdminService.getModelsWithProviders({ includeRemote: false }), + refetchOnWindowFocus: false, + staleTime: 30_000, + }); + + const fullQuery = useQuery({ + queryKey: modelsWithProvidersQueryKey, + queryFn: () => AdminService.getModelsWithProviders(), + refetchOnWindowFocus: false, + staleTime: 60_000, + }); + + const data = fullQuery.data ?? localQuery.data; + + return { + models: data?.models ?? [], + groups: data?.groups ?? [], + isLoading: !data && (localQuery.isLoading || fullQuery.isLoading), + isFetchingRemote: fullQuery.isFetching, + error: data ? null : (fullQuery.error ?? localQuery.error), + refetch: async () => { + await queryClient.invalidateQueries({ + queryKey: modelsWithProvidersQueryKey, + }); + }, + }; +} diff --git a/ui/lib/hooks/use-progressive-list.ts b/ui/lib/hooks/use-progressive-list.ts new file mode 100644 index 00000000..8b55ffc4 --- /dev/null +++ b/ui/lib/hooks/use-progressive-list.ts @@ -0,0 +1,46 @@ +'use client'; + +import { useEffect, useState } from 'react'; + +/** + * Reveal a long list in frame-sized batches. + * + * A provider catalog can hold thousands of rows, and mounting them in one + * commit blocks the main thread long enough that the page looks frozen right + * after navigation. Each batch yields back to the browser, so the first rows + * paint immediately and the rest fill in without freezing input. + */ +export function useProgressiveList( + items: T[], + initialCount = 40, + step = 80 +): { visibleItems: T[]; hiddenCount: number } { + const [count, setCount] = useState(initialCount); + const [trackedItems, setTrackedItems] = useState(items); + + // Reset during render, not in an effect: an effect would first commit the new + // list at the old (possibly full) count, which is the freeze this avoids. + if (trackedItems !== items) { + setTrackedItems(items); + setCount(initialCount); + } + + useEffect(() => { + if (count >= items.length) { + return; + } + + const frame = requestAnimationFrame(() => { + setCount((current) => Math.min(items.length, current + step)); + }); + + return () => cancelAnimationFrame(frame); + }, [count, items.length, step]); + + const visibleCount = Math.min(count, items.length); + + return { + visibleItems: items.slice(0, visibleCount), + hiddenCount: items.length - visibleCount, + }; +} From 1c61dc1bd0af3cabcb5e93bf669ea65944f0be3a Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 02:21:37 +0200 Subject: [PATCH 05/12] refactor: drop provider catalog cache, fetch on page mount with skeleton --- routstr/core/admin.py | 92 +++------------ tests/conftest.py | 15 --- tests/unit/test_admin_remote_models_cache.py | 112 ------------------- ui/components/model-selector.tsx | 30 +---- ui/lib/api/services/admin.ts | 15 +-- ui/lib/hooks/use-models-with-providers.ts | 42 ++----- 6 files changed, 32 insertions(+), 274 deletions(-) delete mode 100644 tests/unit/test_admin_remote_models_cache.py diff --git a/routstr/core/admin.py b/routstr/core/admin.py index f3862edc..365e2f68 100644 --- a/routstr/core/admin.py +++ b/routstr/core/admin.py @@ -1,8 +1,6 @@ -import asyncio import json import re import secrets -import time from datetime import datetime, timezone from pathlib import Path @@ -56,9 +54,6 @@ async def _refresh_provider_model_paths(upstream_provider_id: int) -> None: """Queue discovery sync without blocking the committed admin mutation.""" from ..upstream.model_paths import schedule_model_paths_refresh_for_provider - # Every provider/model mutation funnels through here, so it is also the one - # place that can keep the cached admin catalog from serving a stale listing. - invalidate_remote_models_cache(upstream_provider_id) await schedule_model_paths_refresh_for_provider(upstream_provider_id) @@ -1188,7 +1183,6 @@ async def delete_upstream_provider(provider_id: str) -> dict[str, object]: await session.delete(provider) await session.commit() - invalidate_remote_models_cache(deleted_id) await reinitialize_upstreams() await refresh_model_maps() return {"ok": True, "deleted_id": deleted_id} @@ -1202,78 +1196,13 @@ async def get_provider_types() -> list[dict[str, object]]: return [cls.get_provider_metadata() for cls in upstream_provider_classes] -# The admin catalog view is opened repeatedly and by several panels at once, -# while every miss costs a live upstream round trip. Keep the raw listing for a -# short window and let concurrent readers share one in-flight fetch. -_REMOTE_MODELS_TTL_SECONDS = 120.0 -_REMOTE_MODELS_FETCH_TIMEOUT_SECONDS = 20.0 -_remote_models_cache: dict[int, tuple[float, list]] = {} -_remote_models_locks: dict[int, asyncio.Lock] = {} -# Bumped on every invalidation so a fetch that started against the old provider -# config cannot write its result back after the cache was cleared. -_remote_models_generation = 0 - - -def invalidate_remote_models_cache(provider_pk: int | None = None) -> None: - global _remote_models_generation - _remote_models_generation += 1 - if provider_pk is None: - _remote_models_cache.clear() - _remote_models_locks.clear() - else: - _remote_models_cache.pop(provider_pk, None) - - -async def _get_remote_models( - provider: UpstreamProviderRow, provider_pk: int, force_refresh: bool = False -) -> list: - from ..upstream.helpers import _instantiate_provider - - now = time.monotonic() - cached = _remote_models_cache.get(provider_pk) - if not force_refresh and cached and now - cached[0] < _REMOTE_MODELS_TTL_SECONDS: - return cached[1] - - lock = _remote_models_locks.setdefault(provider_pk, asyncio.Lock()) - async with lock: - cached = _remote_models_cache.get(provider_pk) - now = time.monotonic() - if ( - not force_refresh - and cached - and now - cached[0] < _REMOTE_MODELS_TTL_SECONDS - ): - return cached[1] - - upstream_instance = _instantiate_provider(provider) - if not upstream_instance: - return [] - - generation = _remote_models_generation - try: - models = await asyncio.wait_for( - upstream_instance.fetch_models(), - timeout=_REMOTE_MODELS_FETCH_TIMEOUT_SECONDS, - ) - except Exception as e: - logger.error(f"Failed to fetch models from {provider.provider_type}: {e}") - # A stale listing beats an empty one for an operator view. - return cached[1] if cached else [] - - if generation == _remote_models_generation: - _remote_models_cache[provider_pk] = (time.monotonic(), models) - return models - - @admin_router.get( "/api/upstream-providers/{provider_id}/models", dependencies=[Depends(require_admin_api)], ) -async def get_provider_models( - provider_id: str, - include_remote: bool = Query(True), - refresh_remote: bool = Query(False), -) -> dict[str, object]: +async def get_provider_models(provider_id: str) -> dict[str, object]: + from ..upstream.helpers import _instantiate_provider + async with create_session() as session: provider = await _get_upstream_provider_by_ref(session, provider_id) provider_pk = _provider_pk(provider) @@ -1285,11 +1214,16 @@ async def get_provider_models( apply_fees=False, ) - upstream_models: list = [] - if include_remote: - upstream_models = await _get_remote_models( - provider, provider_pk, force_refresh=refresh_remote - ) + upstream_models = [] + upstream_instance = _instantiate_provider(provider) + if upstream_instance: + try: + raw_models = await upstream_instance.fetch_models() + upstream_models = raw_models + except Exception as e: + logger.error( + f"Failed to fetch models from {provider.provider_type}: {e}" + ) db_model_ids = {model.id for model in db_models} filtered_remote_models = [ diff --git a/tests/conftest.py b/tests/conftest.py index 74c3d058..d1bfa919 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -31,18 +31,3 @@ def _isolate_redemption_negative_cache() -> Iterator[None]: redemption_negative_cache.clear() yield redemption_negative_cache.clear() - - -@pytest.fixture(autouse=True) -def _isolate_admin_remote_models_cache() -> Iterator[None]: - """Clear the admin catalog cache between tests. - - Provider primary keys restart at 1 for every fresh test database, so a - cached listing from an earlier test would otherwise answer for a different - provider that happens to reuse the same key. - """ - from routstr.core.admin import invalidate_remote_models_cache - - invalidate_remote_models_cache() - yield - invalidate_remote_models_cache() diff --git a/tests/unit/test_admin_remote_models_cache.py b/tests/unit/test_admin_remote_models_cache.py deleted file mode 100644 index 6a22ce00..00000000 --- a/tests/unit/test_admin_remote_models_cache.py +++ /dev/null @@ -1,112 +0,0 @@ -"""Cache behavior of the admin provider catalog listing.""" - -import asyncio -from types import SimpleNamespace -from typing import Any -from unittest.mock import AsyncMock, patch - -import pytest - -from routstr.core import admin -from routstr.core.admin import _get_remote_models, invalidate_remote_models_cache - -PROVIDER = SimpleNamespace(provider_type="generic") - - -def _upstream(fetch: Any) -> Any: - return patch( - "routstr.upstream.helpers._instantiate_provider", - return_value=SimpleNamespace(fetch_models=fetch), - ) - - -@pytest.mark.asyncio -async def test_second_read_is_served_from_cache() -> None: - fetch = AsyncMock(return_value=["a"]) - with _upstream(fetch): - assert await _get_remote_models(PROVIDER, 1) == ["a"] # type: ignore[arg-type] - assert await _get_remote_models(PROVIDER, 1) == ["a"] # type: ignore[arg-type] - assert fetch.await_count == 1 - - -@pytest.mark.asyncio -async def test_force_refresh_and_invalidation_refetch() -> None: - fetch = AsyncMock(side_effect=[["a"], ["b"], ["c"]]) - with _upstream(fetch): - await _get_remote_models(PROVIDER, 1) # type: ignore[arg-type] - assert await _get_remote_models(PROVIDER, 1, force_refresh=True) == ["b"] # type: ignore[arg-type] - invalidate_remote_models_cache(1) - assert await _get_remote_models(PROVIDER, 1) == ["c"] # type: ignore[arg-type] - - -@pytest.mark.asyncio -async def test_expired_entry_is_refetched() -> None: - fetch = AsyncMock(side_effect=[["a"], ["b"]]) - with _upstream(fetch): - await _get_remote_models(PROVIDER, 1) # type: ignore[arg-type] - stamp, models = admin._remote_models_cache[1] - admin._remote_models_cache[1] = ( - stamp - admin._REMOTE_MODELS_TTL_SECONDS - 1, - models, - ) - assert await _get_remote_models(PROVIDER, 1) == ["b"] # type: ignore[arg-type] - - -@pytest.mark.asyncio -async def test_failed_refresh_falls_back_to_stale_listing() -> None: - fetch = AsyncMock(side_effect=[["a"], RuntimeError("upstream down")]) - with _upstream(fetch): - await _get_remote_models(PROVIDER, 1) # type: ignore[arg-type] - assert await _get_remote_models(PROVIDER, 1, force_refresh=True) == ["a"] # type: ignore[arg-type] - - -@pytest.mark.asyncio -async def test_failed_first_fetch_returns_empty_and_caches_nothing() -> None: - fetch = AsyncMock(side_effect=RuntimeError("upstream down")) - with _upstream(fetch): - assert await _get_remote_models(PROVIDER, 1) == [] # type: ignore[arg-type] - assert 1 not in admin._remote_models_cache - - -@pytest.mark.asyncio -async def test_concurrent_readers_share_one_fetch() -> None: - release = asyncio.Event() - - async def slow_fetch() -> list[str]: - await release.wait() - return ["a"] - - fetch = AsyncMock(side_effect=slow_fetch) - with _upstream(fetch): - readers = [ - asyncio.create_task(_get_remote_models(PROVIDER, 1)) # type: ignore[arg-type] - for _ in range(5) - ] - await asyncio.sleep(0) - release.set() - results = await asyncio.gather(*readers) - assert results == [["a"]] * 5 - assert fetch.await_count == 1 - - -@pytest.mark.asyncio -async def test_invalidation_during_fetch_discards_in_flight_result() -> None: - release = asyncio.Event() - - async def slow_fetch() -> list[str]: - await release.wait() - return ["old"] - - with _upstream(AsyncMock(side_effect=slow_fetch)): - reader = asyncio.create_task(_get_remote_models(PROVIDER, 1)) # type: ignore[arg-type] - await asyncio.sleep(0) - invalidate_remote_models_cache(1) - release.set() - assert await reader == ["old"] - assert 1 not in admin._remote_models_cache - - -@pytest.mark.asyncio -async def test_uninstantiable_provider_returns_empty() -> None: - with patch("routstr.upstream.helpers._instantiate_provider", return_value=None): - assert await _get_remote_models(PROVIDER, 1) == [] # type: ignore[arg-type] diff --git a/ui/components/model-selector.tsx b/ui/components/model-selector.tsx index 32552ef3..54a345b8 100644 --- a/ui/components/model-selector.tsx +++ b/ui/components/model-selector.tsx @@ -28,7 +28,7 @@ import { AlertDialogHeader, AlertDialogTitle, } from '@/components/ui/alert-dialog'; -import { Trash2, Ban, CheckCircle, Loader2, Plus } from 'lucide-react'; +import { Trash2, Ban, CheckCircle, Plus } from 'lucide-react'; import { toast } from 'sonner'; import { sortModels, @@ -137,7 +137,6 @@ export function ModelSelector({ models, groups, isLoading: isLoadingModels, - isFetchingRemote, error: modelsError, refetch: refetchModels, } = useModelsWithProviders(); @@ -865,19 +864,11 @@ export function ModelSelector({ {Object.keys(groupedModels).length === 0 ? ( - isFetchingRemote ? ( -
- - -
- ) : ( -
-

- Try broadening your search or switch to a different provider - scope. -

-
- ) +
+

+ Try broadening your search or switch to a different provider scope. +

+
) : null} {/* Provider Groups or Filtered Models */} @@ -927,15 +918,6 @@ export function ModelSelector({ ); })} - {/* The stored rows render first; provider catalogs arrive after their - upstream calls return, so the list says more is still on the way. */} - {isFetchingRemote && Object.keys(groupedModels).length > 0 ? ( -
- - Loading provider catalogs… -
- ) : null} - {/* Forms and Dialogs */} {modelDialogState.providerId && ( { - const query = - options.includeRemote === false ? '?include_remote=false' : ''; + static async getProviderModels(providerId: number): Promise { const data = await apiClient.get( - `/admin/api/upstream-providers/${providerId}/models${query}` + `/admin/api/upstream-providers/${providerId}/models` ); // Convert pricing for all models in the list so the UI receives "per 1M tokens" values @@ -433,9 +428,7 @@ export class AdminService { ); } - static async getModelsWithProviders( - options: { includeRemote?: boolean } = {} - ): Promise<{ + static async getModelsWithProviders(): Promise<{ models: AdminModelAsModel[]; groups: AdminModelGroup[]; }> { @@ -460,7 +453,7 @@ export class AdminService { try { return { provider, - models: await this.getProviderModels(provider.id, options), + models: await this.getProviderModels(provider.id), }; } catch (error) { console.error( diff --git a/ui/lib/hooks/use-models-with-providers.ts b/ui/lib/hooks/use-models-with-providers.ts index 90c2f1f5..6aeec5cf 100644 --- a/ui/lib/hooks/use-models-with-providers.ts +++ b/ui/lib/hooks/use-models-with-providers.ts @@ -1,50 +1,26 @@ 'use client'; -import { useQuery, useQueryClient } from '@tanstack/react-query'; +import { useQuery } from '@tanstack/react-query'; import { AdminService } from '@/lib/api/services/admin'; export const modelsWithProvidersQueryKey = ['models-with-providers'] as const; -const localModelsQueryKey = ['models-with-providers', 'local'] as const; /** - * Shared catalog read for every models view. - * - * The database rows come back without touching an upstream, so they render - * first; the listing that needs live provider calls replaces them once it - * lands. Both queries live under one key prefix, so a single - * `invalidateQueries(['models-with-providers'])` still refreshes the pair, and - * every consumer of this hook shares one request instead of fanning out again. + * Shared catalog read for every models view, so the page shell and the + * selector panel share one request instead of each fanning out to providers. */ export function useModelsWithProviders() { - const queryClient = useQueryClient(); - - const localQuery = useQuery({ - queryKey: localModelsQueryKey, - queryFn: () => - AdminService.getModelsWithProviders({ includeRemote: false }), - refetchOnWindowFocus: false, - staleTime: 30_000, - }); - - const fullQuery = useQuery({ + const query = useQuery({ queryKey: modelsWithProvidersQueryKey, queryFn: () => AdminService.getModelsWithProviders(), refetchOnWindowFocus: false, - staleTime: 60_000, }); - const data = fullQuery.data ?? localQuery.data; - return { - models: data?.models ?? [], - groups: data?.groups ?? [], - isLoading: !data && (localQuery.isLoading || fullQuery.isLoading), - isFetchingRemote: fullQuery.isFetching, - error: data ? null : (fullQuery.error ?? localQuery.error), - refetch: async () => { - await queryClient.invalidateQueries({ - queryKey: modelsWithProvidersQueryKey, - }); - }, + models: query.data?.models ?? [], + groups: query.data?.groups ?? [], + isLoading: query.isLoading, + error: query.error, + refetch: query.refetch, }; } From b1799687869d2f86132c00c01051a8fcb12129c8 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 02:27:58 +0200 Subject: [PATCH 06/12] fix: rename params upstreams reject with a named replacement --- routstr/proxy.py | 2 +- routstr/upstream/request_correction.py | 64 ++++++++- tests/unit/test_model_path_routing.py | 69 +++++++++ tests/unit/test_request_correction.py | 190 +++++++++++++++++++++++++ 4 files changed, 323 insertions(+), 2 deletions(-) diff --git a/routstr/proxy.py b/routstr/proxy.py index 38e2dfdc..c585188f 100644 --- a/routstr/proxy.py +++ b/routstr/proxy.py @@ -1015,7 +1015,7 @@ async def _proxy( already_stripped.add(bad_param) logger.warning( "Upstream %s rejected param '%s' for model=%s; " - "stripping and retrying same upstream", + "correcting and retrying same upstream", upstream.provider_type, bad_param, model_id, diff --git a/routstr/upstream/request_correction.py b/routstr/upstream/request_correction.py index d959a015..6bb73632 100644 --- a/routstr/upstream/request_correction.py +++ b/routstr/upstream/request_correction.py @@ -43,6 +43,18 @@ _UNSUPPORTED_PARAM_RE = re.compile( ) +# Matches upstream error text that rejects a param and names its replacement, +# e.g. OpenAI's "Unsupported parameter: 'max_tokens' is not supported with this +# model. Use 'max_completion_tokens' instead." Both names must be quoted so a +# free-form hint like "use gpt-4 instead" never reads as a rename. +_RENAMED_PARAM_RE = re.compile( + r"[`'\"](?P[a-zA-Z_][a-zA-Z0-9_]*)[`'\"]\s+is\s+" + r"(?:deprecated|not\s+supported|unsupported|no\s+longer\s+supported)\b" + r".*?\buse\s+[`'\"](?P[a-zA-Z_][a-zA-Z0-9_]*)[`'\"]\s+instead", + re.IGNORECASE | re.DOTALL, +) + + # A corrector inspects the parsed request body and the upstream error message # and returns ``(new_body_dict, label)`` for a fix it can apply, or ``None`` to # decline. ``label`` identifies the fix so it is applied at most once per request. @@ -100,6 +112,51 @@ _SPEND_SHAPING_PARAMS = frozenset( } ) +# Output caps are interchangeable spellings of the same limit, so moving the +# value from one to another keeps the priced bound intact. +_OUTPUT_CAP_PARAMS = frozenset( + { + "max_tokens", + "max_completion_tokens", + "max_output_tokens", + "max_tokens_to_sample", + } +) + + +def rename_unsupported_param(body: dict, error_message: str) -> tuple[dict, str] | None: + """Move a rejected top-level param to the name the upstream asked for. + + Returns ``(new_body, label)`` with the value carried over unchanged, or + ``None`` when the error names no replacement, the param is absent, or the + replacement is already set. + + A spend-shaping field is only renamed to another output cap: that keeps the + reservation's bound, whereas renaming into or out of any other spend-shaping + field could uncap or fan out the retry. + """ + match = _RENAMED_PARAM_RE.search(error_message) + if not match: + return None + param, replacement = match.group("param"), match.group("replacement") + if param == replacement or param not in body or replacement in body: + return None + param_spend = param.lower() in _SPEND_SHAPING_PARAMS + replacement_spend = replacement.lower() in _SPEND_SHAPING_PARAMS + if (param_spend or replacement_spend) and not ( + param.lower() in _OUTPUT_CAP_PARAMS + and replacement.lower() in _OUTPUT_CAP_PARAMS + ): + logger.warning( + "Upstream asked to rename '%s' to '%s'; refusing because it would " + "change the request's spend bound — surfacing the error", + param, + replacement, + ) + return None + new_body = {(replacement if k == param else k): v for k, v in body.items()} + return new_body, f"{param}->{replacement}" + def strip_unsupported_param(body: dict, error_message: str) -> tuple[dict, str] | None: """Drop a top-level param the upstream named as unsupported/deprecated. @@ -130,7 +187,12 @@ def strip_unsupported_param(body: dict, error_message: str) -> tuple[dict, str] # Ordered pipeline of correctors tried on each recoverable rejection. -DEFAULT_CORRECTORS: tuple[Corrector, ...] = (strip_unsupported_param,) +# Renaming runs first so a param with a named replacement keeps its value +# instead of being dropped. +DEFAULT_CORRECTORS: tuple[Corrector, ...] = ( + rename_unsupported_param, + strip_unsupported_param, +) def correct_request( diff --git a/tests/unit/test_model_path_routing.py b/tests/unit/test_model_path_routing.py index e2c3e52f..cb764289 100644 --- a/tests/unit/test_model_path_routing.py +++ b/tests/unit/test_model_path_routing.py @@ -706,6 +706,75 @@ async def test_pinned_recovery_preserves_routing_fields( fallback.forward_request.assert_not_awaited() +_OPENAI_MAX_TOKENS_ERROR = json.dumps( + { + "error": { + "message": "Unsupported parameter: 'max_tokens' is not supported " + "with this model. Use 'max_completion_tokens' instead.", + "type": "invalid_request_error", + "param": "max_tokens", + "code": "unsupported_parameter", + } + } +).encode() + + +@pytest.mark.asyncio +@pytest.mark.parametrize("pinned", [False, True]) +async def test_rejected_max_tokens_is_renamed_and_retried_on_same_upstream( + pinned: bool, +) -> None: + selected, fallback = _make_upstream(1), _make_upstream(2) + selected.forward_request = AsyncMock( + side_effect=[ + MagicMock(status_code=400, body=_OPENAI_MAX_TOKENS_ERROR), + MagicMock(status_code=200, body=b"{}"), + ] + ) + headers = {"authorization": "Bearer key"} + if pinned: + headers["x-routstr-model-path"] = encode_model_path(selected.base_url, MODEL_ID) + request = _make_request( + headers, + json.dumps( + {"model": MODEL_ID, "max_tokens": 300, "messages": [], "stream": True} + ).encode(), + ) + + response = await _run_proxy( + request, [(MagicMock(), selected), (MagicMock(), fallback)] + ) + + assert response.status_code == 200 + assert selected.forward_request.await_count == 2 + before, after = [ + json.loads(call.args[3]) for call in selected.forward_request.await_args_list + ] + assert before["max_tokens"] == 300 and "max_completion_tokens" not in before + assert after["max_completion_tokens"] == 300 and "max_tokens" not in after + assert {k: v for k, v in after.items() if k != "max_completion_tokens"} == { + k: v for k, v in before.items() if k != "max_tokens" + } + fallback.forward_request.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_rename_that_changes_spend_bound_is_not_retried() -> None: + selected = _make_upstream(1, 400) + selected.forward_request.return_value.body = json.dumps( + {"error": {"message": "'max_tokens' is not supported. Use 'n' instead."}} + ).encode() + request = _make_request( + {"authorization": "Bearer key"}, + json.dumps({"model": MODEL_ID, "max_tokens": 300}).encode(), + ) + + response = await _run_proxy(request, [(MagicMock(), selected)]) + + assert response.status_code == 400 + selected.forward_request.assert_awaited_once() + + # --------------------------------------------------------------------------- # # Upstream 5xx -> 424 + UPSTREAM_UNAVAILABLE + scope header; node faults stay 500. # --------------------------------------------------------------------------- # diff --git a/tests/unit/test_request_correction.py b/tests/unit/test_request_correction.py index 3b1110a2..56a3423c 100644 --- a/tests/unit/test_request_correction.py +++ b/tests/unit/test_request_correction.py @@ -16,9 +16,15 @@ from routstr.upstream.request_correction import ( Correction, correct_request, extract_error_message, + rename_unsupported_param, strip_unsupported_param, ) +OPENAI_MAX_TOKENS_ERROR = ( + "Unsupported parameter: 'max_tokens' is not supported with this model. " + "Use 'max_completion_tokens' instead." +) + def _body(**kwargs: object) -> bytes: return json.dumps(kwargs).encode() @@ -139,6 +145,190 @@ class TestStripUnsupportedParam: assert strip_unsupported_param(body, "`Max_Tokens` is deprecated") is None +class TestRenameUnsupportedParam: + def test_renames_max_tokens_for_openai_reasoning_models(self) -> None: + body = {"model": "gpt-5.6-sol", "max_tokens": 256, "messages": []} + result = rename_unsupported_param(body, OPENAI_MAX_TOKENS_ERROR) + assert result is not None + new_body, label = result + assert label == "max_tokens->max_completion_tokens" + assert new_body == { + "model": "gpt-5.6-sol", + "max_completion_tokens": 256, + "messages": [], + } + + def test_preserves_key_order(self) -> None: + body = {"model": "m", "max_tokens": 1, "stream": True} + result = rename_unsupported_param(body, OPENAI_MAX_TOKENS_ERROR) + assert result is not None + assert list(result[0]) == ["model", "max_completion_tokens", "stream"] + + def test_does_not_mutate_input(self) -> None: + body = {"model": "m", "max_tokens": 8} + assert rename_unsupported_param(body, OPENAI_MAX_TOKENS_ERROR) is not None + assert body == {"model": "m", "max_tokens": 8} + + def test_renames_between_any_output_caps(self) -> None: + caps = ( + "max_tokens", + "max_completion_tokens", + "max_output_tokens", + "max_tokens_to_sample", + ) + for param in caps: + for replacement in caps: + if param == replacement: + continue + message = f"`{param}` is deprecated. Use `{replacement}` instead." + result = rename_unsupported_param({param: 7}, message) + assert result == ({replacement: 7}, f"{param}->{replacement}"), ( + param, + replacement, + ) + + def test_renames_non_spend_param(self) -> None: + message = "'functions' is deprecated. Use 'tools' instead." + result = rename_unsupported_param({"functions": [{"name": "f"}]}, message) + assert result == ({"tools": [{"name": "f"}]}, "functions->tools") + + def test_matches_across_quote_styles_case_and_newlines(self) -> None: + for message in ( + 'Unsupported parameter: "max_tokens" is not supported.\nUse ' + '"max_completion_tokens" instead.', + "`max_tokens` IS UNSUPPORTED here; please USE `max_completion_tokens`" + " INSTEAD", + "'max_tokens' is no longer supported, use 'max_completion_tokens' instead", + ): + result = rename_unsupported_param({"max_tokens": 3}, message) + assert result is not None, message + assert result[0] == {"max_completion_tokens": 3} + + def test_refuses_renames_that_change_the_spend_bound(self) -> None: + for param, replacement in ( + ("max_tokens", "n"), + ("n", "best_of"), + ("best_of", "n"), + ("temperature", "max_tokens"), + ("max_tokens", "temperature"), + ("n", "max_tokens"), + ): + message = f"'{param}' is not supported. Use '{replacement}' instead." + assert rename_unsupported_param({param: 2}, message) is None, ( + param, + replacement, + ) + + def test_spend_guard_is_case_insensitive(self) -> None: + ok = "'Max_Tokens' is not supported. Use 'MAX_COMPLETION_TOKENS' instead." + assert rename_unsupported_param({"Max_Tokens": 4}, ok) == ( + {"MAX_COMPLETION_TOKENS": 4}, + "Max_Tokens->MAX_COMPLETION_TOKENS", + ) + bad = "'Max_Tokens' is not supported. Use 'N' instead." + assert rename_unsupported_param({"Max_Tokens": 4}, bad) is None + + def test_declines_when_replacement_already_present(self) -> None: + body = {"max_tokens": 4, "max_completion_tokens": 8} + assert rename_unsupported_param(body, OPENAI_MAX_TOKENS_ERROR) is None + + def test_declines_when_param_absent(self) -> None: + assert rename_unsupported_param({"model": "m"}, OPENAI_MAX_TOKENS_ERROR) is None + + def test_declines_self_rename(self) -> None: + message = "'max_tokens' is deprecated. Use 'max_tokens' instead." + assert rename_unsupported_param({"max_tokens": 1}, message) is None + + def test_declines_unquoted_or_missing_replacement(self) -> None: + for message in ( + "`gpt-3` is deprecated, use gpt-4 instead", + "'max_tokens' is not supported, use max_completion_tokens instead", + "'max_tokens' is not supported with this model.", + "Use 'max_completion_tokens' instead.", + ): + assert rename_unsupported_param({"max_tokens": 1}, message) is None, message + + def test_declines_nested_only_param(self) -> None: + body = {"reasoning": {"max_tokens": 5}} + assert rename_unsupported_param(body, OPENAI_MAX_TOKENS_ERROR) is None + + +class TestCorrectRequestRename: + def test_openai_max_tokens_error_is_renamed_not_refused(self) -> None: + body = _body(model="gpt-5.6-sol", max_tokens=512, messages=[]) + result = correct_request(body, OPENAI_MAX_TOKENS_ERROR, set()) + assert isinstance(result, Correction) + assert result.label == "max_tokens->max_completion_tokens" + decoded = json.loads(result.body) + assert "max_tokens" not in decoded + assert decoded["max_completion_tokens"] == 512 + + def test_rename_wins_over_strip_for_non_spend_param(self) -> None: + body = _body(model="m", functions=[1]) + result = correct_request( + body, "'functions' is deprecated. Use 'tools' instead.", set() + ) + assert result is not None + assert json.loads(result.body) == {"model": "m", "tools": [1]} + + def test_unsafe_rename_of_cap_still_surfaces_error(self) -> None: + body = _body(model="m", max_tokens=5) + assert ( + correct_request( + body, "'max_tokens' is not supported. Use 'n' instead.", set() + ) + is None + ) + + def test_applied_rename_does_not_repeat_or_strip_cap(self) -> None: + body = _body(model="m", max_tokens=5) + applied = {"max_tokens->max_completion_tokens"} + assert correct_request(body, OPENAI_MAX_TOKENS_ERROR, applied) is None + + def test_rename_ping_pong_terminates(self) -> None: + """An upstream that flip-flops between names cannot loop forever.""" + forward = OPENAI_MAX_TOKENS_ERROR + backward = "'max_completion_tokens' is not supported. Use 'max_tokens' instead." + body = _body(model="m", max_tokens=5) + applied: set[str] = set() + for attempt in range(10): + message = forward if attempt % 2 == 0 else backward + result = correct_request(body, message, applied) + if result is None: + break + body, applied = result.body, applied | {result.label} + else: + raise AssertionError("correction loop did not terminate") + assert applied == { + "max_tokens->max_completion_tokens", + "max_completion_tokens->max_tokens", + } + assert json.loads(body) == {"model": "m", "max_tokens": 5} + + def test_buffered_openai_error_response_is_renamed(self) -> None: + resp = Response( + content=json.dumps( + { + "error": { + "message": OPENAI_MAX_TOKENS_ERROR, + "type": "invalid_request_error", + "param": "max_tokens", + "code": "unsupported_parameter", + } + } + ).encode(), + status_code=400, + ) + body = _body(model="gpt-5.6-sol", max_tokens=64, stream=True) + result = correct_request(body, extract_error_message(resp), set()) + assert result is not None + assert json.loads(result.body) == { + "model": "gpt-5.6-sol", + "max_completion_tokens": 64, + "stream": True, + } + + class TestExtractErrorMessage: def test_extracts_nested_error_message(self) -> None: resp = Response( From 846092473c31a91b6c97403c4b8e872d1b24ecf9 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 03:00:16 +0200 Subject: [PATCH 07/12] fix: send max_completion_tokens to OpenAI reasoning models up front --- routstr/upstream/openai.py | 39 ++++++++++++ tests/unit/test_openai_output_cap.py | 90 ++++++++++++++++++++++++++++ 2 files changed, 129 insertions(+) create mode 100644 tests/unit/test_openai_output_cap.py diff --git a/routstr/upstream/openai.py b/routstr/upstream/openai.py index f2f03cc5..b8d7c923 100644 --- a/routstr/upstream/openai.py +++ b/routstr/upstream/openai.py @@ -1,11 +1,23 @@ +import json from typing import TYPE_CHECKING +from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config +from litellm.llms.openai.chat.o_series_transformation import OpenAIOSeriesConfig + from ..payment.models import Model, async_fetch_openrouter_models from .base import BaseUpstreamProvider if TYPE_CHECKING: from ..core.db import UpstreamProviderRow +_O_SERIES = OpenAIOSeriesConfig() + + +def _rejects_max_tokens(model: str) -> bool: + return OpenAIGPT5Config.is_model_gpt_5_model( + model + ) or _O_SERIES.is_model_o_series_model(model) + class OpenAIUpstreamProvider(BaseUpstreamProvider): """Upstream provider specifically configured for OpenAI API.""" @@ -42,6 +54,33 @@ class OpenAIUpstreamProvider(BaseUpstreamProvider): """Strip 'openai/' prefix for OpenAI API compatibility.""" return model_id.removeprefix("openai/") + def prepare_request_body( + self, + body: bytes | None, + model_obj: Model, + include_stream_usage: bool = False, + ) -> bytes | None: + body = super().prepare_request_body(body, model_obj, include_stream_usage) + if not body: + return body + try: + data = json.loads(body) + except ValueError: + return body + # Reasoning models 400 on max_tokens; renaming up front saves the + # reject-and-retry round trip. Names litellm doesn't know yet still + # fall through to request_correction's reactive rename. + if ( + isinstance(data, dict) + and "messages" in data + and "max_tokens" in data + and "max_completion_tokens" not in data + and _rejects_max_tokens(self.transform_model_name(model_obj.id)) + ): + data["max_completion_tokens"] = data.pop("max_tokens") + return json.dumps(data).encode() + return body + async def fetch_models(self) -> list[Model]: """Fetch OpenAI models from OpenRouter API filtered by openai source.""" models_data = await async_fetch_openrouter_models(source_filter="openai") diff --git a/tests/unit/test_openai_output_cap.py b/tests/unit/test_openai_output_cap.py new file mode 100644 index 00000000..84c7e234 --- /dev/null +++ b/tests/unit/test_openai_output_cap.py @@ -0,0 +1,90 @@ +"""OpenAI reasoning models get ``max_completion_tokens`` before the request is sent.""" + +from __future__ import annotations + +import json +import os + +os.environ.setdefault("UPSTREAM_BASE_URL", "http://test") +os.environ.setdefault("UPSTREAM_API_KEY", "test") +os.environ.setdefault("LIGHTNING_ADDRESS", "test@stm.to") + +import pytest + +from routstr.payment.models import Architecture, Model, Pricing +from routstr.upstream import GenericUpstreamProvider +from routstr.upstream.openai import OpenAIUpstreamProvider + + +def _model(model_id: str) -> Model: + return Model( + id=model_id, + name="test", + created=0, + description="", + context_length=128000, + architecture=Architecture( + modality="text->text", + input_modalities=["text"], + output_modalities=["text"], + tokenizer="x", + instruct_type=None, + ), + pricing=Pricing(prompt=0.0, completion=0.0), + ) + + +def _chat(model_id: str, **fields: object) -> bytes: + return json.dumps( + {"model": model_id, "messages": [{"role": "user", "content": "hi"}], **fields} + ).encode() + + +def _prepare(provider: object, model_id: str, body: bytes) -> dict: + out = provider.prepare_request_body(body, _model(model_id)) # type: ignore[attr-defined] + assert out is not None + return json.loads(out) + + +@pytest.mark.parametrize( + "model_id", ["gpt-5.6-sol", "openai/gpt-6-sol", "openai/gpt-5", "o3", "o4-mini"] +) +def test_reasoning_model_max_tokens_is_renamed(model_id: str) -> None: + provider = OpenAIUpstreamProvider(api_key="k") + data = _prepare(provider, model_id, _chat(model_id, max_tokens=300)) + assert data["max_completion_tokens"] == 300 + assert "max_tokens" not in data + + +@pytest.mark.parametrize("model_id", ["gpt-4o", "openai/gpt-4.1"]) +def test_non_reasoning_model_keeps_max_tokens(model_id: str) -> None: + provider = OpenAIUpstreamProvider(api_key="k") + data = _prepare(provider, model_id, _chat(model_id, max_tokens=300)) + assert data["max_tokens"] == 300 + assert "max_completion_tokens" not in data + + +def test_both_caps_set_is_left_for_upstream() -> None: + provider = OpenAIUpstreamProvider(api_key="k") + data = _prepare( + provider, + "gpt-5.6-sol", + _chat("gpt-5.6-sol", max_tokens=300, max_completion_tokens=200), + ) + assert data["max_tokens"] == 300 + assert data["max_completion_tokens"] == 200 + + +def test_non_chat_body_is_untouched() -> None: + provider = OpenAIUpstreamProvider(api_key="k") + body = json.dumps({"model": "gpt-5.6-sol", "input": "hi", "max_tokens": 5}).encode() + data = _prepare(provider, "gpt-5.6-sol", body) + assert data["max_tokens"] == 5 + assert "max_completion_tokens" not in data + + +def test_other_upstreams_keep_max_tokens() -> None: + provider = GenericUpstreamProvider(base_url="http://test", api_key="k") + data = _prepare(provider, "gpt-5.6-sol", _chat("gpt-5.6-sol", max_tokens=300)) + assert data["max_tokens"] == 300 + assert "max_completion_tokens" not in data From 103596f2b421d3501143f03e77fe86ddab76ea3b Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Wed, 30 Sep 2026 20:01:43 +0800 Subject: [PATCH 08/12] fix: bound billed request lifetimes and recover abandoned reservations --- .../a73d19b6c204_reservation_deadlines.py | 42 ++++++ repro/IMPLEMENTATION.md | 46 ++++++ repro/dummy_upstream.py | 45 ++++++ repro/probe.py | 60 ++++++++ repro/results-final.txt | 19 +++ repro/results.txt | 19 +++ repro/router-final.log | 108 ++++++++++++++ repro/router-first.log | 115 +++++++++++++++ routstr/auth.py | 35 ++++- routstr/core/db.py | 12 +- routstr/core/lifecycle.py | 135 ++++++++++++++++++ routstr/core/main.py | 5 + routstr/core/settings.py | 10 ++ routstr/upstream/stream_ownership.py | 7 +- tests/unit/test_request_lifecycle.py | 57 ++++++++ tests/unit/test_stale_reservations.py | 59 ++++++++ 16 files changed, 770 insertions(+), 4 deletions(-) create mode 100644 migrations/versions/a73d19b6c204_reservation_deadlines.py create mode 100644 repro/IMPLEMENTATION.md create mode 100644 repro/dummy_upstream.py create mode 100644 repro/probe.py create mode 100644 repro/results-final.txt create mode 100644 repro/results.txt create mode 100644 repro/router-final.log create mode 100644 repro/router-first.log create mode 100644 routstr/core/lifecycle.py create mode 100644 tests/unit/test_request_lifecycle.py diff --git a/migrations/versions/a73d19b6c204_reservation_deadlines.py b/migrations/versions/a73d19b6c204_reservation_deadlines.py new file mode 100644 index 00000000..74702d61 --- /dev/null +++ b/migrations/versions/a73d19b6c204_reservation_deadlines.py @@ -0,0 +1,42 @@ +"""Immutable reservation start and absolute recovery deadline. + +Revision ID: a73d19b6c204 +Revises: e4c7a1b9d520 +""" + +import time + +import sqlalchemy as sa +from alembic import op + +revision = "a73d19b6c204" +down_revision = "e4c7a1b9d520" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + op.add_column( + "reservation_releases", sa.Column("started_at", sa.Integer(), nullable=True) + ) + op.add_column( + "reservation_releases", sa.Column("expires_at", sa.Integer(), nullable=True) + ) + op.create_index( + "ix_reservation_releases_expires_at", "reservation_releases", ["expires_at"] + ) + # Original ages are unknowable for renewed legacy rows. Give them a finite + # migration grace period; deploy only after draining old workers. + op.execute( + sa.text( + "UPDATE reservation_releases SET expires_at = :expiry WHERE status = 'active'" + ).bindparams(expiry=int(time.time()) + 1830) + ) + + +def downgrade() -> None: + op.drop_index( + "ix_reservation_releases_expires_at", table_name="reservation_releases" + ) + op.drop_column("reservation_releases", "expires_at") + op.drop_column("reservation_releases", "started_at") diff --git a/repro/IMPLEMENTATION.md b/repro/IMPLEMENTATION.md new file mode 100644 index 00000000..1c231f1f --- /dev/null +++ b/repro/IMPLEMENTATION.md @@ -0,0 +1,46 @@ +# Reservation lifecycle implementation and validation + +Branch: fix/reservation-lifecycle. Baseline: 96c8e2f7. + +## Implemented + +- Outermost pure-ASGI lifecycle supervision with one coordinated receive consumer, explicit disconnect monitoring, cancellation, and exact reservation fallback cleanup. +- Finite overall request lifetime (MAX_REQUEST_LIFETIME_SECONDS, default 1800), downstream send timeout (DOWNSTREAM_SEND_TIMEOUT_SECONDS, default 60), and cleanup timeout (REQUEST_CLEANUP_TIMEOUT_SECONDS, default 30). +- Lifecycle identity shared through context across middleware tasks; reservation replacements are registered for exact cleanup. +- Heartbeats stop on lifecycle termination or local maximum age. +- Persistent stream finalization has a finite cleanup budget. +- Durable immutable started_at and expires_at columns; expiry covers remaining request lifetime plus settlement grace, including provider fallback without restarting the original deadline. +- Renewal and charge claims refuse expired reservations. Sweeping can release absolute-expired reservations even when their renewable timestamp is fresh. +- Migration grants legacy active rows 1830 seconds of grace; original ages are not fabricated. Drain old workers before deployment. + +## Verification + +Run from worktree with PYTHONPATH=$PWD because the shared root virtual environment's editable install points at the original checkout: + +PYTHONPATH=$PWD ../../.venv/bin/pytest tests/unit/test_request_lifecycle.py tests/unit/test_stale_reservations.py tests/unit/test_streaming_billing_finalization.py tests/integration/test_negative_available_balance_repro.py -q + +64 tests passed. Ruff checks passed on changed files. Full-project mypy was attempted but did not finish within the tool timeout; no successful typecheck is claimed. + +Final built image: localhost/routstr-reserved-repro:fix, ef81426ad79e3d14ec462a39ab1f7481fd0cb410a9cb93cb42de41e8b3523869. + +Container tests used real TCP, full middleware stack, frozen image dependencies, isolated SQLite and synthetic balances. Read timeout 3s, lifetime 15s, delivery timeout 2s, cleanup timeout 3s, stale timeout 6s. + +Reused the main probe on ports 18100/18101. Results in results-final.txt and router-final.log: + +- Finite and silent streams settled. +- Header wait released its reservation. +- Disconnected endless stream no longer retained its reservation. +- Non-reading flood client hit bounded delivery/cleanup. +- Connected keepalive-only stream terminated at maximum lifetime. +- After the background-sweep interval and all client closures: every key reserved_balance=0, no active durable reservations. Explicit database assertions passed. +- Router shut down within the 10-second grace without SIGKILL. Dummy upstream still required SIGKILL: its fixture deliberately sleeps/open-streams and is not patched router code. + +Actual mint payout was not tested. Protocol errors on already-started streams when deadlines interrupt them are expected; an HTTP status cannot be replaced after headers are sent. + +## Financial policy / limitations + +The lifecycle first lets existing finalization run within a bounded budget. If still active, fallback releases only that reservation; late charge is fenced by terminal state. This can forgo charging observed output on failed settlement. It prioritizes freeing customer funds over leaving them locked; review this policy before deployment. Upstream compute may continue remotely even after local connection closure. + +This implementation does not complete every proposed hardening idea: provider cancellation APIs, full observability, per-record unexpected DB-failure isolation, legacy NULL aggregate background reconciliation, multi-worker/alternate-route network matrix and DB-outage injection remain follow-up work. No dependency upgrade was needed for the tested cases because explicit disconnect supervision avoids relying solely on send errors. + +All reproduction containers are stopped. Original node data/configuration is untouched. Source changes are uncommitted in the worktree for review. diff --git a/repro/dummy_upstream.py b/repro/dummy_upstream.py new file mode 100644 index 00000000..1be75ecc --- /dev/null +++ b/repro/dummy_upstream.py @@ -0,0 +1,45 @@ +"""Loopback-only streaming fixture; no router monkeypatches.""" +import asyncio +import json +import time +from fastapi import FastAPI, Request +from fastapi.responses import StreamingResponse + +app = FastAPI() +events = [] + +@app.get('/events') +async def history(): + return events + +@app.get('/v1/models') +async def models(): + return {'object': 'list', 'data': [{'id': 'gpt-4o-mini', 'object': 'model', 'created': 1, 'owned_by': 'repro'}]} + +@app.post('/v1/chat/completions') +async def completions(request: Request): + body = await request.json() + mode = body.get('messages', [{}])[0].get('content', 'finite') + events.append({'event': 'start', 'mode': mode, 'time': time.time()}) + if mode.startswith('header'): + await asyncio.sleep(3600) + async def stream(): + count = 0 + try: + while True: + if mode.startswith('keepalive'): + yield ': ping\n\n' + else: + chunk = {'id': 'repro', 'object': 'chat.completion.chunk', 'created': int(time.time()), 'model': 'gpt-4o-mini', 'choices': [{'index': 0, 'delta': {'content': 'x' * (65536 if mode.startswith('flood') else 1)}, 'finish_reason': None}]} + yield 'data: ' + json.dumps(chunk) + '\n\n' + count += 1 + if mode == 'finite' and count >= 3: + yield 'data: ' + json.dumps({'id': 'repro', 'object': 'chat.completion.chunk', 'model': 'gpt-4o-mini', 'choices': [], 'usage': {'prompt_tokens': 1, 'completion_tokens': count, 'total_tokens': count + 1}}) + '\n\n' + yield 'data: [DONE]\n\n' + return + await asyncio.sleep(3600 if mode.startswith('silent') else (0.001 if mode.startswith('flood') else 0.5)) + finally: + event = {'event': 'close', 'mode': mode, 'chunks': count, 'time': time.time()} + events.append(event) + print(json.dumps(event), flush=True) + return StreamingResponse(stream(), media_type='text/event-stream') diff --git a/repro/probe.py b/repro/probe.py new file mode 100644 index 00000000..93d71a75 --- /dev/null +++ b/repro/probe.py @@ -0,0 +1,60 @@ +import asyncio +import json +import socket +import subprocess +import time +import httpx + +BASE='http://127.0.0.1:18100' + +def snapshot(): + code="import sqlite3,json,time; c=sqlite3.connect('/tmp/reserved-fix.db'); c.row_factory=sqlite3.Row; print(json.dumps({'time':time.time(),'keys':[dict(r) for r in c.execute(\"select hashed_key,balance,reserved_balance,reserved_at from api_keys where hashed_key like 'main-%'\")],'rows':[dict(r) for r in c.execute(\"select * from reservation_releases where key_hash like 'main-%'\")]}))" + return json.loads(subprocess.check_output(['podman','exec','reserved-router-fix','/.venv/bin/python','-c',code],text=True)) + +async def consume(mode): + try: + async with httpx.AsyncClient(timeout=None) as c: + async with c.stream('POST',BASE+'/v1/chat/completions',headers={'Authorization':'Bearer sk-main-'+mode},json={'model':'gpt-4o-mini','messages':[{'role':'user','content':mode}],'stream':True,'max_tokens':10}) as r: + print('STREAM',mode,r.status_code,flush=True) + async for _ in r.aiter_bytes(): pass + print('ENDED',mode,flush=True) + except asyncio.CancelledError: + print('CLIENT_DISCONNECTED',mode,flush=True) + raise + except Exception as e: + print('CLIENT_ERROR',mode,type(e).__name__,str(e),flush=True) + +async def report(label): + print(label,json.dumps(snapshot()),flush=True) + async with httpx.AsyncClient(timeout=5) as c: + for mode in ['silent-disconnect','endless-disconnect','keepalive','flood','header']: + # Only attempt payout while reserved: avoid requiring a real mint. + if next(k for k in snapshot()['keys'] if k['hashed_key']=='main-'+mode)['reserved_balance']: + r=await c.post(BASE+'/v1/wallet/refund',headers={'Authorization':'Bearer sk-main-'+mode}) + print('REFUND',mode,r.status_code,r.text,flush=True) + print('UPSTREAM_EVENTS',json.dumps((await c.get('http://127.0.0.1:18101/events')).json()),flush=True) + +async def main(): + modes=['finite','silent','silent-disconnect','endless-disconnect','keepalive','header'] + tasks={m:asyncio.create_task(consume(m)) for m in modes} + # Real client with a small receive buffer, never draining the HTTP response. + sock=socket.socket(); sock.setsockopt(socket.SOL_SOCKET,socket.SO_RCVBUF,1024); sock.connect(('127.0.0.1',18100)) + body=json.dumps({'model':'gpt-4o-mini','messages':[{'role':'user','content':'flood'}],'stream':True,'max_tokens':10}).encode() + sock.sendall(b'POST /v1/chat/completions HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer sk-main-flood\r\nContent-Type: application/json\r\nContent-Length: '+str(len(body)).encode()+b'\r\n\r\n'+body) + await asyncio.sleep(1) + for m in ['silent-disconnect','endless-disconnect']: + tasks[m].cancel() + await asyncio.gather(tasks['silent-disconnect'],tasks['endless-disconnect'],return_exceptions=True) + await asyncio.sleep(9) + await report('AT_10_SECONDS') + await asyncio.sleep(60) + await report('AFTER_SWEEP') + sock.close() + tasks['keepalive'].cancel() + await asyncio.gather(tasks['keepalive'],return_exceptions=True) + await asyncio.sleep(8) + await report('AFTER_ALL_CLIENTS_CLOSED') + for task in tasks.values(): task.cancel() + await asyncio.gather(*tasks.values(),return_exceptions=True) + +asyncio.run(main()) diff --git a/repro/results-final.txt b/repro/results-final.txt new file mode 100644 index 00000000..6545bf6d --- /dev/null +++ b/repro/results-final.txt @@ -0,0 +1,19 @@ +STREAM endless-disconnect 200 +STREAM keepalive 200 +STREAM finite 200 +STREAM silent 200 +STREAM silent-disconnect 200 +CLIENT_DISCONNECTED silent-disconnect +CLIENT_DISCONNECTED endless-disconnect +ENDED finite +ENDED silent +STREAM header 424 +ENDED header +AT_10_SECONDS {"time": 1790767971.1792026, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 12, "reserved_at": 1790767960}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "9b5b9063f6f045a290611e4144c897de", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "released", "created_at": 1790767962, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "efc8563301ba458890ee3ab2335e4e49", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cbed89b7ec444fea9789cf53b3e0f476", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "active", "created_at": 1790767971, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1c56b25265df4743b4cff70dc57544c6", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cc7f113eef004b7ba27bc761c5d9b9a1", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "6a1b5506b8ab4747884475075ebc38da", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767961, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1dfedc1c2cee407fb2cd1a28b1253c6d", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}]} +REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"45f783cc-4c0b-4222-be13-8e726fc7cebc"} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}, {"event": "start", "mode": "flood", "time": 1790767960.9594278}, {"event": "start", "mode": "endless-disconnect", "time": 1790767961.008998}, {"event": "start", "mode": "keepalive", "time": 1790767961.036595}, {"event": "start", "mode": "finite", "time": 1790767961.060875}, {"event": "start", "mode": "silent", "time": 1790767961.0869172}, {"event": "start", "mode": "silent-disconnect", "time": 1790767961.1580715}, {"event": "start", "mode": "header", "time": 1790767961.1944675}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767962.064207}] +CLIENT_ERROR keepalive RemoteProtocolError peer closed connection without sending complete message body (incomplete chunked read) +AFTER_SWEEP {"time": 1790768033.9280283, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "9b5b9063f6f045a290611e4144c897de", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "released", "created_at": 1790767962, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "efc8563301ba458890ee3ab2335e4e49", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cbed89b7ec444fea9789cf53b3e0f476", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767975, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1c56b25265df4743b4cff70dc57544c6", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cc7f113eef004b7ba27bc761c5d9b9a1", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "6a1b5506b8ab4747884475075ebc38da", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767961, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1dfedc1c2cee407fb2cd1a28b1253c6d", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}]} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}, {"event": "start", "mode": "flood", "time": 1790767960.9594278}, {"event": "start", "mode": "endless-disconnect", "time": 1790767961.008998}, {"event": "start", "mode": "keepalive", "time": 1790767961.036595}, {"event": "start", "mode": "finite", "time": 1790767961.060875}, {"event": "start", "mode": "silent", "time": 1790767961.0869172}, {"event": "start", "mode": "silent-disconnect", "time": 1790767961.1580715}, {"event": "start", "mode": "header", "time": 1790767961.1944675}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767962.064207}] +AFTER_ALL_CLIENTS_CLOSED {"time": 1790768044.521246, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "9b5b9063f6f045a290611e4144c897de", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "released", "created_at": 1790767962, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "efc8563301ba458890ee3ab2335e4e49", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cbed89b7ec444fea9789cf53b3e0f476", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767975, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1c56b25265df4743b4cff70dc57544c6", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cc7f113eef004b7ba27bc761c5d9b9a1", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "6a1b5506b8ab4747884475075ebc38da", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767961, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1dfedc1c2cee407fb2cd1a28b1253c6d", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}]} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}, {"event": "start", "mode": "flood", "time": 1790767960.9594278}, {"event": "start", "mode": "endless-disconnect", "time": 1790767961.008998}, {"event": "start", "mode": "keepalive", "time": 1790767961.036595}, {"event": "start", "mode": "finite", "time": 1790767961.060875}, {"event": "start", "mode": "silent", "time": 1790767961.0869172}, {"event": "start", "mode": "silent-disconnect", "time": 1790767961.1580715}, {"event": "start", "mode": "header", "time": 1790767961.1944675}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767962.064207}] diff --git a/repro/results.txt b/repro/results.txt new file mode 100644 index 00000000..01417f00 --- /dev/null +++ b/repro/results.txt @@ -0,0 +1,19 @@ +STREAM silent-disconnect 200 +STREAM keepalive 200 +STREAM endless-disconnect 200 +STREAM silent 200 +STREAM finite 200 +CLIENT_DISCONNECTED silent-disconnect +CLIENT_DISCONNECTED endless-disconnect +ENDED finite +STREAM header 424 +ENDED header +ENDED silent +AT_10_SECONDS {"time": 1790767727.5333533, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 12, "reserved_at": 1790767717}, {"hashed_key": "main-flood", "balance": 999893370, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "79c7feb2274140748da2a97180f56d2c", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ecf20c01870b4ce49fc81bd300ee35df", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ec6f854d6a8647ac8b3bba50752bc647", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "active", "created_at": 1790767725, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "6da67d4c5cbd4ab19cc30f2f0fac6aaa", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "f4a35c5e89c343bfbe21415dce28a4d5", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "50a37cffba7745fa84d03b4070c86066", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "63ca23fa71024428baaf8baeb7ee3ede", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}]} +REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"f0bcc404-dffe-4860-8091-308f721ba053"} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}] +CLIENT_ERROR keepalive RemoteProtocolError peer closed connection without sending complete message body (incomplete chunked read) +AFTER_SWEEP {"time": 1790767790.5300848, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 999893370, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "79c7feb2274140748da2a97180f56d2c", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ecf20c01870b4ce49fc81bd300ee35df", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ec6f854d6a8647ac8b3bba50752bc647", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767731, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "6da67d4c5cbd4ab19cc30f2f0fac6aaa", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "f4a35c5e89c343bfbe21415dce28a4d5", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "50a37cffba7745fa84d03b4070c86066", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "63ca23fa71024428baaf8baeb7ee3ede", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}]} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}] +AFTER_ALL_CLIENTS_CLOSED {"time": 1790767800.920076, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 999893370, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "79c7feb2274140748da2a97180f56d2c", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ecf20c01870b4ce49fc81bd300ee35df", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ec6f854d6a8647ac8b3bba50752bc647", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767731, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "6da67d4c5cbd4ab19cc30f2f0fac6aaa", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "f4a35c5e89c343bfbe21415dce28a4d5", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "50a37cffba7745fa84d03b4070c86066", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "63ca23fa71024428baaf8baeb7ee3ede", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}]} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}] diff --git a/repro/router-final.log b/repro/router-final.log new file mode 100644 index 00000000..dee5fe67 --- /dev/null +++ b/repro/router-final.log @@ -0,0 +1,108 @@ +/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block + return result +2026-09-30 11:32:06 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). +2026-09-30 11:32:06 INFO uvicorn.error Started server process [1] +2026-09-30 11:32:06 INFO uvicorn.error Waiting for application startup. +2026-09-30 11:32:06 INFO routstr.core.main Application startup initiated +2026-09-30 11:32:10 INFO routstr.core.db Database migrations completed successfully +2026-09-30 11:32:10 INFO routstr.core.db Reset reserved balances on startup +2026-09-30 11:32:11 INFO routstr.upstream.helpers Seeding custom provider +2026-09-30 11:32:11 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings +2026-09-30 11:32:12 INFO routstr.proxy Initialized 1 upstream providers +2026-09-30 11:32:12 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider +2026-09-30 11:32:12 INFO routstr.nostr.analytics Usage analytics sharing task started +2026-09-30 11:32:12 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr +2026-09-30 11:32:12 INFO routstr.auth Dead-key pruning disabled (interval <= 0) +2026-09-30 11:32:12 INFO uvicorn.error Application startup complete. +2026-09-30 11:32:12 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18100 (Press CTRL+C to quit) +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found +2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:32:40 INFO routstr.auth Processing payment for request +2026-09-30 11:32:40 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:40 INFO routstr.payments RESERVE +2026-09-30 11:32:40 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:40 INFO routstr.payments RESERVE +2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:41 INFO routstr.payments RESERVE +2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:41 INFO routstr.payments RESERVE +2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:41 INFO routstr.payments RESERVE +2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:41 INFO routstr.payments RESERVE +2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully +2026-09-30 11:32:41 INFO routstr.payments RESERVE +2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:32:41 INFO routstr.auth Payment settlement finished +2026-09-30 11:32:41 INFO routstr.auth Payment settlement finished +2026-09-30 11:32:42 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:32:42 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:32:42 INFO routstr.auth Calculated token-based cost +2026-09-30 11:32:42 INFO routstr.auth Refunding excess payment +2026-09-30 11:32:42 INFO routstr.auth Refund processed successfully +2026-09-30 11:32:42 INFO routstr.payments FINALIZE +2026-09-30 11:32:42 INFO routstr.auth Payment settlement finished +2026-09-30 11:32:42 INFO routstr.upstream.auto_topup Auto top-up worker started +2026-09-30 11:32:43 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:32:43 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:32:43 ERROR routstr.core.exceptions Unhandled exception +asyncio.exceptions.CancelledError + +The above exception was the direct cause of the following exception: + +TimeoutError +2026-09-30 11:32:43 ERROR uvicorn.error Exception in ASGI application +asyncio.exceptions.CancelledError + +The above exception was the direct cause of the following exception: + +TimeoutError +2026-09-30 11:32:43 INFO routstr.auth Payment settlement finished +2026-09-30 11:32:44 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream +2026-09-30 11:32:44 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:32:44 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:32:44 INFO routstr.auth Calculated token-based cost +2026-09-30 11:32:44 INFO routstr.auth Refunding excess payment +2026-09-30 11:32:44 INFO routstr.auth Refund processed successfully +2026-09-30 11:32:44 INFO routstr.payments FINALIZE +2026-09-30 11:32:44 INFO routstr.auth Payment settlement finished +2026-09-30 11:32:44 ERROR routstr.core.exceptions Unhandled exception +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:32:44 ERROR uvicorn.error Exception in ASGI application +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:32:44 ERROR routstr.upstream.base HTTP request error to upstream +2026-09-30 11:32:44 WARNING routstr.proxy Upstream base failed for model=gpt-4o-mini: Upstream service request timed out +2026-09-30 11:32:52 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:32:55 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:32:55 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:32:55 ERROR uvicorn.error ASGI callable returned without completing response. +2026-09-30 11:32:55 INFO routstr.auth Payment settlement finished diff --git a/repro/router-first.log b/repro/router-first.log new file mode 100644 index 00000000..b8979fc0 --- /dev/null +++ b/repro/router-first.log @@ -0,0 +1,115 @@ +/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block + return result +2026-09-30 11:28:16 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). +2026-09-30 11:28:16 INFO uvicorn.error Started server process [1] +2026-09-30 11:28:16 INFO uvicorn.error Waiting for application startup. +2026-09-30 11:28:16 INFO routstr.core.main Application startup initiated +2026-09-30 11:28:20 INFO routstr.core.db Database migrations completed successfully +2026-09-30 11:28:21 INFO routstr.core.db Reset reserved balances on startup +2026-09-30 11:28:21 INFO routstr.upstream.helpers Seeding custom provider +2026-09-30 11:28:21 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings +2026-09-30 11:28:22 INFO routstr.proxy Initialized 1 upstream providers +2026-09-30 11:28:22 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider +2026-09-30 11:28:22 INFO routstr.nostr.analytics Usage analytics sharing task started +2026-09-30 11:28:22 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr +2026-09-30 11:28:22 INFO routstr.auth Dead-key pruning disabled (interval <= 0) +2026-09-30 11:28:22 INFO uvicorn.error Application startup complete. +2026-09-30 11:28:22 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18100 (Press CTRL+C to quit) +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:28:37 INFO routstr.auth Processing payment for request +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:28:37 INFO routstr.payments RESERVE +2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:28:38 INFO routstr.auth Payment settlement finished +2026-09-30 11:28:38 INFO routstr.auth Calculated token-based cost +2026-09-30 11:28:38 INFO routstr.auth Refunding excess payment +2026-09-30 11:28:38 INFO routstr.auth Refund processed successfully +2026-09-30 11:28:38 INFO routstr.payments FINALIZE +2026-09-30 11:28:38 INFO routstr.auth Payment settlement finished +2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:28:38 INFO routstr.auth Calculated token-based cost +2026-09-30 11:28:38 INFO routstr.auth Refunding excess payment +2026-09-30 11:28:38 INFO routstr.auth Refund processed successfully +2026-09-30 11:28:38 INFO routstr.payments FINALIZE +2026-09-30 11:28:38 INFO routstr.auth Payment settlement finished +2026-09-30 11:28:39 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:28:39 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:28:39 INFO routstr.auth Calculated token-based cost +2026-09-30 11:28:39 INFO routstr.auth Finalized payment with additional charge +2026-09-30 11:28:39 INFO routstr.payments FINALIZE +2026-09-30 11:28:39 INFO routstr.auth Payment settlement finished +2026-09-30 11:28:39 ERROR routstr.core.exceptions Unhandled exception +asyncio.exceptions.CancelledError + +The above exception was the direct cause of the following exception: + +TimeoutError +2026-09-30 11:28:39 ERROR uvicorn.error Exception in ASGI application +asyncio.exceptions.CancelledError + +The above exception was the direct cause of the following exception: + +TimeoutError +2026-09-30 11:28:40 ERROR routstr.upstream.base HTTP request error to upstream +2026-09-30 11:28:40 WARNING routstr.proxy Upstream base failed for model=gpt-4o-mini: Upstream service request timed out +2026-09-30 11:28:40 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream +2026-09-30 11:28:40 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:28:40 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:28:40 INFO routstr.auth Calculated token-based cost +2026-09-30 11:28:40 INFO routstr.auth Refunding excess payment +2026-09-30 11:28:40 INFO routstr.auth Refund processed successfully +2026-09-30 11:28:40 INFO routstr.payments FINALIZE +2026-09-30 11:28:40 INFO routstr.auth Payment settlement finished +2026-09-30 11:28:40 ERROR routstr.core.exceptions Unhandled exception +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:28:40 ERROR uvicorn.error Exception in ASGI application +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:28:49 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:28:52 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:28:52 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:28:52 ERROR uvicorn.error ASGI callable returned without completing response. +2026-09-30 11:28:52 INFO routstr.auth Payment settlement finished +2026-09-30 11:28:52 INFO routstr.upstream.auto_topup Auto top-up worker started diff --git a/routstr/auth.py b/routstr/auth.py index 9bfc7b3d..b7c8e16a 100644 --- a/routstr/auth.py +++ b/routstr/auth.py @@ -637,6 +637,14 @@ async def pay_for_request( ) # Charge the base cost for the request atomically to avoid race conditions + from .core.lifecycle import request_lifetime + + lifetime = request_lifetime.get() + remaining_lifetime = ( + max(0, lifetime.deadline - asyncio.get_running_loop().time()) + if lifetime is not None + else settings.max_request_lifetime_seconds + ) reserved_at_now = int(time.time()) stmt = ( update(ApiKey) @@ -686,6 +694,9 @@ async def pay_for_request( billing_key_hash=reservation.billing_key_hash, reserved_msats=reservation.reserved_msats, status="active", + started_at=reserved_at_now, + expires_at=reserved_at_now + + math.ceil(remaining_lifetime + settings.request_cleanup_timeout_seconds), ) ) # Publish the identity before commit. If the commit succeeds but its @@ -726,6 +737,11 @@ async def pay_for_request( # The reservation is durable; keep its lease fresh for the whole request # lifetime (upstream header waits, non-streaming and streaming alike). + from .core.lifecycle import request_lifetime + + lifetime = request_lifetime.get() + if lifetime is not None: + lifetime.reservations.append(reservation) _start_reservation_heartbeat(reservation) try: @@ -875,6 +891,10 @@ async def renew_reservation( update(ReservationRelease) .where(col(ReservationRelease.id) == snapshot.release_id) .where(col(ReservationRelease.status) == "active") + .where( + (col(ReservationRelease.expires_at).is_(None)) + | (col(ReservationRelease.expires_at) > int(time.time())) + ) .values(created_at=int(time.time())) ) await session.commit() @@ -902,12 +922,21 @@ def _start_reservation_heartbeat(snapshot: ReservationSnapshot) -> None: """ interval = max(1, settings.stale_reservation_timeout_seconds // 3) owner = asyncio.current_task() + from .core.lifecycle import request_lifetime + + lifetime = request_lifetime.get() + deadline = asyncio.get_running_loop().time() + settings.max_request_lifetime_seconds async def beat() -> None: try: while True: await asyncio.sleep(interval) - if owner is None or owner.done(): + if ( + owner is None + or owner.done() + or (lifetime is not None and lifetime.stopped) + or asyncio.get_running_loop().time() >= deadline + ): # Request control is gone; let the lease expire so the # sweeper can release the reservation if no terminal # transition ever ran. @@ -1091,6 +1120,10 @@ async def _claim_reservation_for_charge( update(ReservationRelease) .where(col(ReservationRelease.id) == snapshot.release_id) .where(col(ReservationRelease.status) == "active") + .where( + col(ReservationRelease.expires_at).is_(None) + | (col(ReservationRelease.expires_at) > int(time.time())) + ) .where(col(ReservationRelease.key_hash) == snapshot.key_hash) .where(col(ReservationRelease.billing_key_hash) == snapshot.billing_key_hash) .where(col(ReservationRelease.reserved_msats) == snapshot.reserved_msats) diff --git a/routstr/core/db.py b/routstr/core/db.py index 500420d5..84307a7e 100644 --- a/routstr/core/db.py +++ b/routstr/core/db.py @@ -174,7 +174,10 @@ async def _transition_stale_reservation( update(ReservationRelease) .where(col(ReservationRelease.id) == reservation_id) .where(col(ReservationRelease.status) == "active") - .where(col(ReservationRelease.created_at) < cutoff) + .where( + (col(ReservationRelease.created_at) < cutoff) + | (col(ReservationRelease.expires_at) <= int(time.time())) + ) .values(status="released") ) return bool(transition.rowcount == 1) @@ -221,7 +224,10 @@ async def release_stale_reservations( query = ( select(ReservationRelease) .where(col(ReservationRelease.status) == "active") - .where(col(ReservationRelease.created_at) < cutoff) + .where( + (col(ReservationRelease.created_at) < cutoff) + | (col(ReservationRelease.expires_at) <= int(time.time())) + ) ) if key_hash is not None: query = query.where( @@ -784,6 +790,8 @@ class ReservationRelease(SQLModel, table=True): # type: ignore key_hash: str = Field(index=True) billing_key_hash: str = Field(index=True) reserved_msats: int + started_at: int | None = Field(default=None) + expires_at: int | None = Field(default=None, index=True) status: str = Field(default="active") created_at: int = Field(default_factory=lambda: int(time.time())) diff --git a/routstr/core/lifecycle.py b/routstr/core/lifecycle.py new file mode 100644 index 00000000..897df5b7 --- /dev/null +++ b/routstr/core/lifecycle.py @@ -0,0 +1,135 @@ +"""Supervise the real downstream connection, outside HTTP middleware wrappers.""" + +from __future__ import annotations + +import asyncio +from contextvars import ContextVar +from dataclasses import dataclass, field +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from ..auth import ReservationSnapshot + +from starlette.types import ASGIApp, Message, Receive, Scope, Send + +from . import get_logger +from .settings import settings + +logger = get_logger(__name__) + + +@dataclass +class RequestLifetime: + deadline: float = 0 + stopped: bool = False + reservations: list[ReservationSnapshot] = field(default_factory=list) + + +request_lifetime: ContextVar[RequestLifetime | None] = ContextVar( + "request_lifetime", default=None +) + + +class RequestLifecycleMiddleware: + def __init__(self, app: ASGIApp) -> None: + self.app = app + + async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: + if scope["type"] != "http": + await self.app(scope, receive, send) + return + lifetime = RequestLifetime( + deadline=asyncio.get_running_loop().time() + + settings.max_request_lifetime_seconds + ) + token = request_lifetime.set(lifetime) + disconnected = asyncio.Event() + # One receive consumer. Backpressure uploads until consumed; after the + # final body message, continue listening independently of the app. + messages: asyncio.Queue[Message] = asyncio.Queue(maxsize=1) + response_started = False + + async def pump() -> None: + while True: + message = await receive() + if message["type"] == "http.disconnect": + disconnected.set() + return + await messages.put(message) + + async def downstream_receive() -> Message: + if disconnected.is_set(): + return {"type": "http.disconnect"} + get = asyncio.create_task(messages.get()) + gone = asyncio.create_task(disconnected.wait()) + try: + await asyncio.wait((get, gone), return_when=asyncio.FIRST_COMPLETED) + if disconnected.is_set(): + return {"type": "http.disconnect"} + return get.result() + finally: + for task in (get, gone): + task.cancel() + await asyncio.gather(get, gone, return_exceptions=True) + + async def downstream_send(message: Message) -> None: + nonlocal response_started + if disconnected.is_set() or lifetime.stopped: + raise OSError("Downstream request terminated") + async with asyncio.timeout(settings.downstream_send_timeout_seconds): + await send(message) + if message["type"] == "http.response.start": + response_started = True + + receiver = asyncio.create_task(pump()) + work = asyncio.create_task(self.app(scope, downstream_receive, downstream_send)) + gone = asyncio.create_task(disconnected.wait()) + try: + done, _ = await asyncio.wait( + (work, gone), + timeout=settings.max_request_lifetime_seconds, + return_when=asyncio.FIRST_COMPLETED, + ) + if work in done: + await work + elif not disconnected.is_set() and not response_started: + await downstream_send( + {"type": "http.response.start", "status": 504, "headers": []} + ) + await downstream_send( + {"type": "http.response.body", "body": b"Request deadline exceeded"} + ) + finally: + lifetime.stopped = True + for task in (receiver, gone, work): + task.cancel() + # Cancellation/close is bounded: an uncooperative finalizer must not + # hold ownership or renewal indefinitely. + done, pending = await asyncio.wait( + (receiver, gone, work), timeout=settings.request_cleanup_timeout_seconds + ) + for task in done: + if not task.cancelled(): + task.exception() + for task in pending: + task.cancel() + task.add_done_callback( + lambda t: t.exception() if not t.cancelled() else None + ) + try: + async with asyncio.timeout(settings.request_cleanup_timeout_seconds): + from ..auth import _stop_reservation_heartbeat, release_reservation + from .db import create_session + + for snapshot in lifetime.reservations: + await _stop_reservation_heartbeat(snapshot.release_id) + async with create_session() as session: + await release_reservation( + snapshot, session, snapshot.reserved_msats + ) + except Exception: + logger.exception( + "Request cleanup failed; durable expiry will recover reservations" + ) + finally: + request_lifetime.reset(token) diff --git a/routstr/core/main.py b/routstr/core/main.py index 584fa736..46aad1fc 100644 --- a/routstr/core/main.py +++ b/routstr/core/main.py @@ -45,6 +45,7 @@ from .exceptions import ( http_exception_handler, validation_exception_handler, ) +from .lifecycle import RequestLifecycleMiddleware from .logging import get_logger, setup_logging from .middleware import LoggingMiddleware from .not_found import _NOT_FOUND_HTML, not_found_catch_all # noqa: F401 @@ -315,6 +316,10 @@ app.add_middleware( # Add logging middleware app.add_middleware(LoggingMiddleware) +# Outermost: observe the actual downstream connection, not middleware streams. + +app.add_middleware(RequestLifecycleMiddleware) + # Add exception handlers app.add_exception_handler(HTTPException, http_exception_handler) # type: ignore app.add_exception_handler(RequestValidationError, validation_exception_handler) diff --git a/routstr/core/settings.py b/routstr/core/settings.py index 5dd2d57d..d9324f13 100644 --- a/routstr/core/settings.py +++ b/routstr/core/settings.py @@ -117,6 +117,16 @@ class Settings(BaseSettings): default=604_800, env="DEAD_KEY_MIN_AGE_SECONDS" ) + max_request_lifetime_seconds: float = Field( + default=1800, gt=0, env="MAX_REQUEST_LIFETIME_SECONDS" + ) + downstream_send_timeout_seconds: float = Field( + default=60, gt=0, env="DOWNSTREAM_SEND_TIMEOUT_SECONDS" + ) + request_cleanup_timeout_seconds: float = Field( + default=30, gt=0, env="REQUEST_CLEANUP_TIMEOUT_SECONDS" + ) + # Network cors_origins: list[str] = Field(default_factory=lambda: ["*"], env="CORS_ORIGINS") # Comma-separated METHOD:path pairs adding to the proxy's canonical diff --git a/routstr/upstream/stream_ownership.py b/routstr/upstream/stream_ownership.py index e0e3e51d..ea06e400 100644 --- a/routstr/upstream/stream_ownership.py +++ b/routstr/upstream/stream_ownership.py @@ -10,6 +10,7 @@ from fastapi.responses import StreamingResponse from starlette.types import Receive, Scope, Send from ..core import get_logger +from ..core.settings import settings logger = get_logger(__name__) @@ -74,10 +75,14 @@ class PersistentStreamFinalizer: self._task: asyncio.Future[None] | None = None self._lock = asyncio.Lock() + async def _bounded_finalize(self) -> None: + async with asyncio.timeout(settings.request_cleanup_timeout_seconds): + await self._finalize() + async def run(self) -> None: async with self._lock: if self._task is None: - self._task = asyncio.ensure_future(self._finalize()) + self._task = asyncio.ensure_future(self._bounded_finalize()) task = self._task await asyncio.shield(task) diff --git a/tests/unit/test_request_lifecycle.py b/tests/unit/test_request_lifecycle.py new file mode 100644 index 00000000..03cf9897 --- /dev/null +++ b/tests/unit/test_request_lifecycle.py @@ -0,0 +1,57 @@ +import asyncio +from unittest.mock import patch + +import pytest + +from routstr.core.lifecycle import RequestLifecycleMiddleware +from routstr.core.settings import settings + + +@pytest.mark.asyncio +@pytest.mark.parametrize("reason", ["disconnect", "deadline", "send"]) +async def test_lifecycle_stops_live_work(reason): + closed = asyncio.Event() + receive_queue = asyncio.Queue() + await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) + sent = [] + + async def app(scope, receive, send): + try: + assert (await receive())["type"] == "http.request" + await send({"type": "http.response.start", "status": 200, "headers": []}) + while True: + await send( + {"type": "http.response.body", "body": b"x", "more_body": True} + ) + await asyncio.sleep(0.01) + finally: + closed.set() + + async def send(message): + sent.append(message) + if reason == "send" and message["type"] == "http.response.body": + await asyncio.sleep(100) + + async def disconnect(): + await asyncio.sleep(0.02) + await receive_queue.put({"type": "http.disconnect"}) + + task = asyncio.create_task(disconnect()) if reason == "disconnect" else None + with ( + patch.object(settings, "max_request_lifetime_seconds", 0.08), + patch.object(settings, "downstream_send_timeout_seconds", 0.03), + patch.object(settings, "request_cleanup_timeout_seconds", 0.1), + ): + try: + await asyncio.wait_for( + RequestLifecycleMiddleware(app)( + {"type": "http"}, receive_queue.get, send + ), + 1, + ) + except TimeoutError: + assert reason == "send" + if task: + await task + assert closed.is_set() + assert sent diff --git a/tests/unit/test_stale_reservations.py b/tests/unit/test_stale_reservations.py index 558fdda5..6cd59682 100644 --- a/tests/unit/test_stale_reservations.py +++ b/tests/unit/test_stale_reservations.py @@ -428,3 +428,62 @@ async def test_proxy_reverts_reservation_on_client_disconnect() -> None: await proxy_module.proxy(request, "v1/chat/completions") revert_mock.assert_awaited_once_with(key, session, 1000, reservation_snapshot) + + +@pytest.mark.asyncio +async def test_absolute_expiry_releases_fresh_lease(session: AsyncSession) -> None: + now = int(time.time()) + key = ApiKey( + hashed_key="expired-deadline", + balance=5000, + reserved_balance=1000, + reserved_at=now, + ) + session.add(key) + session.add( + ReservationRelease( + id="expired", + key_hash=key.hashed_key, + billing_key_hash=key.hashed_key, + reserved_msats=1000, + created_at=now, + started_at=now - 100, + expires_at=now - 1, + ) + ) + await session.commit() + assert await release_stale_reservations(session, 300) == 1 + await session.refresh(key) + assert key.reserved_balance == 0 + assert key.balance == 5000 + + +@pytest.mark.asyncio +async def test_expired_reservation_cannot_renew_or_claim_charge( + session: AsyncSession, +) -> None: + from routstr.auth import ( + ReservationSnapshot, + _claim_reservation_for_charge, + renew_reservation, + ) + + snapshot = ReservationSnapshot( + release_id="fenced", + key_hash="fenced-key", + billing_key_hash="fenced-key", + reserved_msats=1000, + ) + session.add(ApiKey(hashed_key="fenced-key", balance=5000, reserved_balance=1000)) + session.add( + ReservationRelease( + id="fenced", + key_hash="fenced-key", + billing_key_hash="fenced-key", + reserved_msats=1000, + expires_at=int(time.time()) - 1, + ) + ) + await session.commit() + assert not await renew_reservation(snapshot, session) + assert not await _claim_reservation_for_charge(snapshot, session) From 8aa86b8814eedfe0564f3d2c9e372a45ec4989bf Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Wed, 30 Sep 2026 20:03:24 +0800 Subject: [PATCH 09/12] docs: include reservation diagnosis and original main reproduction with fix --- RESERVED_BALANCE.md | 541 ++++++++++++++++++ reservation-repro-main/README.md | 106 ++++ .../after-upstream-stop.json | 1 + reservation-repro-main/connections.txt | 7 + reservation-repro-main/control-results.txt | 5 + reservation-repro-main/control-router.log | 43 ++ reservation-repro-main/dummy_upstream.py | 45 ++ reservation-repro-main/final-before-stop.json | 1 + reservation-repro-main/no_logging_app.py | 4 + reservation-repro-main/probe.py | 60 ++ reservation-repro-main/results.txt | 27 + reservation-repro-main/router.log | 114 ++++ reservation-repro-main/starlette-source.txt | 169 ++++++ reservation-repro-main/upstream.log | 23 + reservation-repro-main/uvicorn-source.txt | 125 ++++ 15 files changed, 1271 insertions(+) create mode 100644 RESERVED_BALANCE.md create mode 100644 reservation-repro-main/README.md create mode 100644 reservation-repro-main/after-upstream-stop.json create mode 100644 reservation-repro-main/connections.txt create mode 100644 reservation-repro-main/control-results.txt create mode 100644 reservation-repro-main/control-router.log create mode 100644 reservation-repro-main/dummy_upstream.py create mode 100644 reservation-repro-main/final-before-stop.json create mode 100644 reservation-repro-main/no_logging_app.py create mode 100644 reservation-repro-main/probe.py create mode 100644 reservation-repro-main/results.txt create mode 100644 reservation-repro-main/router.log create mode 100644 reservation-repro-main/starlette-source.txt create mode 100644 reservation-repro-main/upstream.log create mode 100644 reservation-repro-main/uvicorn-source.txt diff --git a/RESERVED_BALANCE.md b/RESERVED_BALANCE.md new file mode 100644 index 00000000..a7417bc4 --- /dev/null +++ b/RESERVED_BALANCE.md @@ -0,0 +1,541 @@ +# Reserved balance blocks refunds long after the last request + +## Reported issue + +A client attempting to refund an API key receives: + +> Cannot refund key. There are ongoing requests for this api key. + +The user reports that the key has not been used in a very long time, potentially days. This is not a refund racing with normal request completion. The expected behavior is that reservations left by disconnected, crashed, abandoned, or failed requests eventually expire and the key becomes refundable. + +The error does **not** prove that an upstream inference request is running. In the current implementation, it means the refund endpoint still sees a positive aggregate `reserved_balance` after attempting stale-reservation cleanup. + +This document records a source-code investigation of the current checkout. The affected node's database, logs, runtime tasks, effective configuration, and deployed version have not been inspected. The production root cause remains unconfirmed. + +## Investigation scope and results + +Checkout inspected: `96c8e2f7` (`Merge pull request #790 from Routstr/fix/rename-unsupported-param`). + +The existing cleanup system is implemented and wired into application startup. It protects several important accounting invariants, but it is based on renewable reservation leases rather than a hard maximum request lifetime. + +Verification command: + +```bash +.venv/bin/pytest \ + tests/unit/test_stale_reservations.py \ + tests/unit/test_streaming_billing_finalization.py \ + tests/integration/test_negative_available_balance_repro.py -q +``` + +Result: **59 passed in 10.04 seconds**. + +These passing tests verify existing recovery paths; they do not establish what happened on the affected node or demonstrate recovery from every kind of live-but-hung task. No implementation changes were made during this investigation. + +## Reservation lifecycle + +### 1. Reserve before forwarding + +`pay_for_request()` in `routstr/auth.py` reserves funds before dispatching the billed request upstream. + +It creates a durable `ReservationRelease` identity containing: + +- `id`: the individual reservation identity; +- `key_hash`: the request's key; +- `billing_key_hash`: the key whose balance backs the request; +- `reserved_msats`: the amount owned by this reservation; +- `status`: initially `active`; +- `created_at`: initially the current timestamp. + +The aggregate reserved balance and durable reservation row commit together. The request's reservation identity matters: releasing one request must not erase funds reserved by another concurrent request. + +`ApiKey.reserved_at` is also stamped when funds are reserved. It is an aggregate timestamp, not an independent timestamp for each request. + +### 2. Renew while the owner task remains alive + +`_start_reservation_heartbeat()` in `routstr/auth.py` starts a task for each reservation. Its interval is: + +```python +max(1, settings.stale_reservation_timeout_seconds // 3) +``` + +With the default timeout of 300 seconds, renewal occurs approximately every 100 seconds. + +The heartbeat captures `asyncio.current_task()` as the owner. At each iteration it checks: + +```python +if owner is None or owner.done(): + return +``` + +If the owner is still alive, it calls `renew_reservation()` using a separate database session. Renewal updates the active durable row's `created_at` to the current time. + +Important consequences: + +- Renewal depends on task lifetime, not demonstrated request progress. +- There is no original-age limit in this heartbeat. +- `created_at` is overwritten, so it actually serves as a renewable lease timestamp. +- An owner that has finished cannot keep renewing indefinitely through this heartbeat. +- An owner that is blocked indefinitely may keep renewing indefinitely. + +### 3. Settle or release + +Normal completion settles the charge and releases the reservation. Handled upstream failures revert the reservation. Terminal reservation transitions stop the heartbeat. + +The proxy includes cancellation cleanup. Streaming paths use finalizers and ownership wrappers to improve cleanup across cancellation and downstream-send failures. Relevant code includes: + +- `routstr/auth.py`; +- `routstr/proxy.py`; +- `routstr/upstream/base.py`; +- `routstr/upstream/stream_ownership.py`. + +If a request dies without completing cleanup, its heartbeat is intended to stop once the owning task is done. The reservation can then age out and be released by the sweeper. + +## Existing cleanup mechanisms + +### Background sweep + +`periodic_stale_reservation_sweep()` in `routstr/auth.py` is started by the application lifespan in `routstr/core/main.py`. + +Defaults: + +| Setting/mechanism | Default | Meaning | +| --- | --- | --- | +| `STALE_RESERVATION_TIMEOUT_SECONDS` | 300 seconds | Maximum age of an unrenewed reservation lease before it is stale | +| `STALE_RESERVATION_SWEEP_INTERVAL_SECONDS` | 60 seconds | Interval between background cleanup passes | +| Heartbeat interval | 100 seconds | Approximately one third of the stale timeout | +| `UPSTREAM_READ_TIMEOUT` | 900 seconds | Upstream HTTP read inactivity timeout, not a total request deadline | +| `RESET_RESERVED_BALANCE_ON_STARTUP` | `True` | Explicit startup reset of active reservations and aggregate reserved balances | + +The sweeper calls `release_stale_reservations()` in `routstr/core/db.py`. + +For durable reservations, it selects `active` rows whose `created_at` is older than the cutoff. Its terminal update also checks the timestamp, protecting against a heartbeat that renews between selection and release. + +Each successful release subtracts that reservation's own amount from the relevant aggregates. Healthy releases commit individually so that certain later corruption repairs cannot roll them back. + +Under healthy execution, recovery occurs after the last lease renewal has aged beyond the configured timeout, plus sweep scheduling and database-operation time. This is **not** a guarantee of release 300 seconds after the request originally began. + +### Refund-time cleanup + +`refund_wallet_endpoint()` in `routstr/balance.py` checks for reserved funds before opening the refund claim. + +If `key.reserved_balance > 0`, it: + +1. Calls `release_stale_reservations()` scoped to that key. +2. Refreshes the key from the database. +3. Returns HTTP 400 with the reported message if reserved balance remains. + +Thus, the current refund path does not rely exclusively on the background task having run. A stale durable reservation should also be releasable during refund itself. + +If cleanup raises an unexpected exception instead, that is a separate failure from this specific HTTP 400 branch. + +### Legacy aggregate cleanup + +Older deployments may have aggregate reserved balances without matching durable rows. + +The cleanup function also looks for these legacy aggregates, but only clears them when there is no active durable owner. It uses a compare-and-swap guard on the observed balance and timestamp to avoid erasing a newly created reservation. + +The behavior differs between background and targeted cleanup: + +| Legacy aggregate state, with no active durable owner | Background sweep | Refund-time targeted cleanup | +| --- | --- | --- | +| Old `reserved_at` | Eligible for release | Eligible for release | +| Recent `reserved_at` | Preserved | Preserved | +| `reserved_at = NULL` | Deliberately skipped | Eligible for repair | + +The NULL-timestamp behavior is explicitly covered by existing tests. It is a background-recovery limitation, but **alone it does not explain the reported refund rejection on the current checkout**, because targeted refund cleanup heals it. + +### Startup reset + +When enabled, startup calls `reset_all_reserved_balances()`. It marks active durable reservations released and clears aggregate reserved balances and timestamps. + +This is not a safe universal operational fix. In a shared-database, multi-instance setup, another instance may still own a legitimate in-flight request. Resetting its reservation can break billing. The setting's source comment recommends disabling it for horizontal scaling. + +## Why the 900-second HTTP timeout does not guarantee eventual completion + +The user correctly asks: if the last request was days ago, shouldn't a 900-second upstream timeout have completed or failed the request long before now? + +**For an ordinary request actively waiting for upstream bytes, with no bytes arriving, yes.** It should hit the read timeout and reach failure cleanup. A days-long refund blockage is abnormal, not expected behavior for a silent upstream. + +However, the HTTP read timeout is not an absolute deadline spanning the complete request lifecycle. + +### Upstream continues sending bytes + +A stream can avoid a read inactivity timeout by delivering bytes periodically. Those bytes might be content or keepalive traffic. A stream with no total-duration limit could therefore remain open longer than 900 seconds. + +This is a technical possibility, **not evidence that the affected upstream streamed for days**. It must not be assumed as the production explanation. + +### Router is blocked writing to the downstream client + +If the router has received a chunk and is blocked delivering it to the client, it may not currently be waiting on an upstream HTTP read. The upstream read timeout is not a general bound on downstream ASGI sends. + +Whether a particular blocked send keeps the captured owner task alive depends on the execution path. That behavior needs a runtime trace or regression test, rather than an assumption about all stream paths. + +### Router is blocked after upstream completion + +Database settlement, finalization, or resource cleanup happens outside the upstream read operation. The upstream read timeout does not bound these waits. + +If the heartbeat's owning task remains alive while waiting, renewal may continue. If that owner finishes and only detached cleanup remains, the heartbeat should stop and the sweeper should eventually recover the reservation. + +### Conclusion + +The current code has no common hard lifetime limit found in this investigation that covers reservation creation, upstream dispatch, streaming delivery, and finalization together. + +The missing guarantee is: + +> A live-but-stuck request cannot renew its reservation forever. + +This gap is confirmed by the heartbeat's renewal condition. The specific blocked operation, if any, on the affected node is not known. + +## Findings and hypotheses + +### Confirmed: renewal does not require progress + +An owner task being alive is sufficient to renew the lease. Neither original request age nor meaningful progress is checked. + +This permits indefinite reservation retention in principle, even without new requests using the key. + +### Confirmed: immutable request age is not stored in the reservation row + +`ReservationRelease.created_at` doubles as the last-renewal timestamp. Once renewed, it cannot tell us when the request originally started. + +This impairs diagnostics and prevents enforcing an original-age limit from this field alone. + +### Confirmed: NULL legacy timestamps are not background-cleaned + +Such keys may remain reserved indefinitely in the background. The current refund endpoint has targeted recovery for this state, subject to the absence of an active durable owner. + +### Confirmed: unexpected failures can interrupt a sweep pass + +The background loop catches unexpected exceptions, logs `Error in periodic_stale_reservation_sweep`, and retries after the sweep interval. + +Some aggregate-corruption cases are handled per reservation, but not every database exception is isolated per record. A persistently failing operation could repeatedly interrupt a pass. Whether this prevents a particular key's cleanup depends on the failure and processing order. + +There is no evidence yet that this caused the reported error. + +### Possible: affected deployment differs from this checkout + +The current code includes heartbeat-owner binding, targeted legacy recovery, and corruption handling. The affected node may run older or different code. + +The deployed commit must be established before treating local behavior as proof of production behavior. + +### Possible: future timestamps or unusual effective configuration + +A future-dated lease can remain non-stale unexpectedly. An unusually large configured timeout can also preserve old reservations. + +Clock skew between instances sharing a database can affect lease timestamps and age calculations. These are diagnostic checks, not confirmed causes. + +## Existing verified recovery coverage + +The suites run during this investigation cover, among other cases: + +- Stamping aggregate reservation timestamps on payment. +- Reverting individual reservations without erasing siblings. +- Releasing old reservations and preserving fresh ones. +- Resetting reserved balances during explicit startup reset. +- Refund-time recovery of stale and legacy NULL-timestamp aggregates. +- Refusing refunds while a recent reservation remains. +- Streaming finalization and client-disconnect cleanup. +- Owner task termination allowing recovery of an abandoned reservation. +- Lease renewal across an in-flight request. +- Renewal racing with stale release. +- Legacy aggregate release racing with a new reservation. +- Several accounting-corruption cases and safe terminal repair. +- Preventing late charges after a reservation has reached a released terminal state. + +These tests do not substitute for explicit tests of endless keepalive streams, blocked downstream sends, or finalization that never completes. + +## Production diagnosis: distinguish a renewing lease from failed cleanup + +The most useful initial question is: + +> Is the reservation still being renewed, or is it stale and not being released? + +Do not share the raw API-key secret. Use its stored hash and reservation identifiers in restricted operational diagnostics. + +### 1. Establish deployment and configuration + +Record: + +- Deployed commit/version. +- Effective `STALE_RESERVATION_TIMEOUT_SECONDS`. +- Effective `UPSTREAM_READ_TIMEOUT`. +- Startup-reset setting. +- Number of instances sharing the database. +- Current time on each relevant instance. +- Whether the lifespan/background tasks completed startup. + +Use effective settings, not only environment variables; settings initialization includes persisted configuration. + +### 2. Inspect the key and all related reservations + +Read-only queries: + +```sql +SELECT hashed_key, balance, reserved_balance, reserved_at +FROM api_keys +WHERE hashed_key = :key_hash; + +SELECT id, key_hash, billing_key_hash, + reserved_msats, status, created_at +FROM reservation_releases +WHERE key_hash = :key_hash + OR billing_key_hash = :key_hash; +``` + +Inspect both key relationships, since a reservation may reference the key as request owner or billing owner. + +Take two snapshots approximately 110 seconds apart with default settings, or use an interval longer than the effective heartbeat interval. A pair of snapshots is a useful signal; it is not a substitute for longer observation when renewal is delayed or intermittent. + +### 3. Interpret the results + +| Observation | Investigation direction | +| --- | --- | +| Active reservation timestamp advances | Identify the instance and owning task renewing it; inspect its stack and actual progress | +| Active reservation timestamp is older than the stale cutoff and does not advance | Check sweep execution/errors, refund cleanup, deployed code, and accounting state | +| Reserved balance remains with no active durable rows | Inspect legacy timestamp and aggregate recovery; current targeted refund cleanup should repair stale/NULL state | +| Lease timestamp is in the future | Check clocks and timestamp integrity | +| Some rows are stale and others fresh | Release only stale owners; do not clear the whole key | +| Aggregate amount disagrees with active durable ownership | Investigate accounting drift and safe reconciliation | + +If the lease is genuinely days old and unrenewed, the indefinite-heartbeat explanation does **not** explain that row. Cleanup failure or incompatible deployment becomes the relevant direction. + +### 4. Inspect logs and task state + +Relevant existing log messages include: + +- `Error in periodic_stale_reservation_sweep`. +- `Failed to renew billing reservation lease`. +- `Released stale reservations`. +- `Released corrupt stale reservation without aggregate subtraction`. +- `Released corrupt reservation without aggregate subtraction`. +- `Client disconnected mid-request, reverting reservation`. +- `refund_wallet_endpoint: released stale reservation before refund`. + +For a renewing lease, locate the process with that reservation's heartbeat and inspect the owner's stack. Determine whether it is waiting on upstream input, downstream delivery, database work, finalization, or another operation. + +Also correlate the original request with upstream outcome and billing logs. A heartbeat alone does not demonstrate that inference is still running. + +## Proposed hardening + +These are proposed changes, not completed fixes. + +### 1. Separate original age from renewable lease age + +Keep distinct durable fields for: + +- Immutable reservation/request start time. +- Last lease renewal time. + +Consider additional progress and ownership metadata where justified. Define migration behavior explicitly: existing renewed `created_at` values cannot reconstruct true original start times. + +### 2. Bound the actual request, not just the accounting lease + +Introduce a configurable total billed-request lifetime covering all relevant routes and phases, including streaming delivery. Add appropriate inactivity bounds for upstream waits and downstream delivery, and bounded finalization/cleanup behavior. + +Timeout handling should: + +1. Stop or cancel the owning request and close owned resources. +2. Settle known or estimated delivered usage according to existing billing policy. +3. Release only that request's remaining reservation. +4. Stop heartbeat renewal. +5. Reach a durable terminal state that prevents later charging. + +Do **not** merely stop renewal or zero the key while a request continues running. Releasing funds while upstream work can still finish creates refund/late-charge and provider-cost risks. + +Care is also needed not to cancel legitimate long-running inference accidentally. Request lifetime, inactivity, and lease expiry are different concepts and should have distinct documented policies. + +### 3. Improve stalled-owner detection and observability + +Expose actionable, non-secret diagnostics: + +- Reservation identity and owning instance. +- Immutable age and current lease age. +- Last meaningful progress and current phase, if tracked. +- Reason for terminal transition or refused refund. +- Age and count of active reservations. +- Sweep failures and cleanup duration. + +Do not treat upstream keepalive bytes as necessarily meaningful model progress. Decide deliberately which signals should extend which deadlines. + +### 4. Reconcile legacy and inconsistent aggregates safely + +Define a migration/recovery policy for NULL legacy timestamps, rather than leaving them background-ineligible indefinitely. + +Mixed-version deployments require caution: an aggregate without a durable row might still belong to an older live worker. Any reconciliation must preserve valid durable owners and avoid unsafe whole-key resets. + +Investigate positive residual aggregates even after durable rows become terminal, with concurrency guards and accounting invariants preserved. + +### 5. Make cleanup failures diagnosable and resilient + +Consider bounded database operations, per-record failure isolation where safe, and alerts for repeated sweep failures or reservations exceeding expected age. + +Failure isolation must not weaken atomicity between durable transitions and aggregate updates. A failed release must not partially debit unrelated reservations. + +## Regression tests needed to close the gaps + +Add tests that reproduce and verify recovery for: + +1. A live owner waiting indefinitely without progress. +2. An endless upstream stream sending keepalive bytes below the read-timeout interval. +3. A downstream send blocked indefinitely after receiving an upstream chunk. +4. Finalization or database settlement that stalls. +5. Cancellation before streaming begins, during streaming, and during finalization. +6. Renewing lease older than the new maximum original-age limit. +7. Background legacy NULL-timestamp recovery under the chosen migration policy. +8. Corrupt residual aggregates alongside a healthy active sibling reservation. +9. A failing cleanup operation followed by other recoverable reservations. +10. Multiple workers concurrently renewing, sweeping, timing out, and refunding. +11. Late completion attempting to charge after timeout/release. +12. Future lease timestamps and the chosen clock-skew policy. + +For each timeout/recovery test, assert: + +- The underlying request/resource is stopped or closed as intended. +- No heartbeat can renew indefinitely afterward. +- Only the affected reservation is released. +- Sibling reservations remain intact. +- Balance/reserved accounting remains valid. +- Terminal transitions are idempotent. +- A later completion cannot charge released/refunded funds. +- The key becomes refundable when no legitimate reservations remain. + +## Operational caution + +Do not solve the symptom by manually setting `reserved_balance = 0` while active requests or heartbeat tasks may exist. Durable reservation state and aggregate balances must agree, and late completion must not be allowed to spend refunded funds. + +Any production repair should begin with a read-only snapshot and identification of live ownership, then use an accounting-safe terminal transition or controlled maintenance procedure. + +## Bottom line + +The expected stale cleanup exists. A genuinely dead, unrenewed reservation should recover on the current version with healthy database access, including during a refund attempt. + +The confirmed design gap is that **a task remaining alive is sufficient to renew its reservation indefinitely**, and the upstream 900-second read timeout does not bound every phase of that task's lifetime. + +A days-old refund blockage therefore warrants investigation, not an assumption that normal request processing is still underway. The first decisive evidence is whether the affected reservation's lease timestamp continues advancing. The production root cause and implementation fixes remain open. + +## Release-specific reproduction: v0.4.7 (confirmed) + +The user subsequently confirmed that the affected node runs the released **v0.4.7** tag. Testing that tag revealed an important correction to the initial analysis above: + +**The 900-second upstream read timeout exists in the newer checkout, not in v0.4.7.** The release's forwarding paths construct `httpx.AsyncClient(..., timeout=None)`. It has no `upstream_read_timeout` settings field. Setting `UPSTREAM_READ_TIMEOUT=3` in the reproduction did nothing; importing the release settings confirmed the field is absent. + +Therefore, on this release an upstream can send one chunk and then remain completely silent without triggering an HTTP read timeout. Periodic bytes are not needed to explain indefinite waiting. + +### Environment and isolation + +- Podman: 5.8.4, netavark network backend. +- Release commit: `f32565e2547abbbffd77a01198ef683ecb8e3d4f`. +- Detached worktree: `.worktrees/reserved-balance-v047`. +- Built the release's own Dockerfile (Python 3.11 base), without source patches. +- Image: `localhost/routstr-reserved-repro:v0.4.7`. +- Image ID: `8c3340a33040df37439f7085e369050d3acc2adf0324bc024e1b4288a3601f76`. +- Separate containers, loopback ports 18080/18081, container-local SQLite database. +- No original node database, wallet, secrets, volumes, or image tag were changed. +- Host networking avoided the reported aardvark DNS issue for this experiment; containerized DNS was not tested or repaired. +- Accelerated stale timeout: 6 seconds, heartbeat every 2 seconds. Background sweep retained its actual 60-second interval. +- Synthetic database-funded keys avoided introducing Cashu mint behavior into the reservation test. Actual refund payout success was not tested. + +### Dummy upstream scenarios + +A small local OpenAI-compatible server exposed `/v1/models` and `/v1/chat/completions` using `gpt-4o-mini`: + +1. **Finite:** three chunks, a usage event, and `[DONE]`. +2. **Silent:** one chunk, then sleep for 3600 seconds. +3. **Endless:** a content chunk every 0.5 seconds with no terminal event. + +The test client consumed streams, queried reservation state, attempted refunds, and disconnected. Evidence and reusable scripts are in `reservation-repro-v047/`. + +### Observed results + +| Scenario | Outcome | +| --- | --- | +| Finite stream | Settled normally; reserved balance became zero | +| Silent stream | Did not time out; durable lease kept renewing | +| Endless stream | Lease kept renewing; refund returned the exact reported HTTP 400 | +| Both clients disconnected | Both upstream connections remained established; both reservations remained active and kept renewing | +| After more than a background-sweep interval | The abandoned reservations were still active; their fresh leases prevented stale cleanup | +| Dummy upstream forcibly stopped | Both requests finally reached error/finalization; both reserved balances became zero and rows became `charged` | + +Both streams reserved 11 msats. Their lease timestamps initially advanced from `1790765677` through `1790765685` and `1790765695`. After client termination, a later snapshot at `1790765791` still showed both rows `active` with leases at `1790765789`. This is approximately 116 seconds after their creation and well beyond the accelerated stale timeout and a background-sweep interval. + +At that later point, `ss` showed two established router-to-upstream connections and no test-client connection on port 18080. A refund for the silent key still returned: + +```json +{"detail":"Cannot refund key. There are ongoing requests for this api key."} +``` + +Stopping the dummy upstream broke those connections. Finalization then charged estimated usage and cleared the reservations. The finite and silent keys ended with a 3-msat charge; the endless stream accumulated a 25-msat charge. This also demonstrates that abandoned upstream work can continue affecting billing after the downstream client is gone. + +The first probe run ended with a client-side `TimeoutError` because it expected the silent stream to complete. That timeout was imposed by the probe's `asyncio.wait_for`, not by the router. The saved probe was subsequently adjusted to report this expected observation rather than crash. + +### What this establishes + +We have reproduced a plausible mechanism for a key remaining blocked long after the client last used it on **the exact release tag**: + +1. The upstream stream remains open, even silently. +2. Downstream disconnection does not terminate the upstream-owning request in the tested runtime/path. +3. The owner remains alive, so its heartbeat keeps renewing. +4. Background and refund-time stale cleanup preserve the fresh lease. +5. Refund remains blocked indefinitely unless the upstream closes or another intervention stops the owning work. + +The reproduction lasted minutes, not days. The absence of a read timeout and continuing renewal explain how the state can persist longer; no days-long run was performed. + +This is concrete release-specific evidence, but not proof that the affected production key has this exact state. Production confirmation still requires reservation snapshots and logs. + +### Shutdown symptoms + +The dummy upstream also needed SIGKILL after a short SIGTERM grace period while its streams were open. The router stopped normally after the upstream was stopped and its streams finalized. + +This supports the possibility that outstanding streaming work can delay graceful shutdown. It does not establish that the user's earlier router/UI shutdown warnings share the same cause. The aardvark DNS removal failure is a separate Podman networking symptom; the reproduction does not require it. + +### Next implementation work + +Prioritize fixes/backports appropriate to v0.4.7: + +- Finite upstream transport timeouts, including reads and header waits. +- Reliable downstream-disconnect propagation and deterministic closure/finalization of owned streaming resources in the deployed FastAPI/Starlette/Uvicorn combination. +- A maximum request lifetime independent of renewable leases and keepalive bytes. +- Real-network regression tests that disconnect a client from a silent upstream stream and assert upstream closure, terminal billing state, stopped renewal, and zero residual reservation. + +The newer checkout has transport timeout and stream-ownership changes, but this experiment did not validate the same scenario against that newer checkout. Do not assume an upgrade fully fixes every gap without rerunning the reproduction. + +Both reproduction containers were stopped at the end. Their container-local database and logs were retained for inspection; no original services were restarted. + +## Current main reproduction: timeout does not close every gap + +The same investigation was repeated against unpatched local main commit `96c8e2f77de8e9f8a0979d17dba0a6d20c78fe89` using its own Dockerfile and frozen dependencies. The main image ran Python 3.14, Starlette 1.6.0, and Uvicorn 0.31.1. Detailed commands and evidence are in `reservation-repro-main/README.md`. + +With an effective upstream read timeout of 3 seconds and stale timeout of 6 seconds: + +- Finite completion settled correctly. +- Silent upstream streams reached the read timeout and cleared reservations. +- A header wait timed out with HTTP 424 and released its reservation. +- An endless content stream **after client disconnect** kept renewing and returning the reported refund HTTP 400. +- SSE comment-only keepalives evaded read timeout; renewal persisted even after disconnect. +- A flood stream to a downstream client that never read remained reserved, including after its socket closed. + +The three problematic keys remained active approximately 269 seconds after request start, across multiple background sweeps, with fresh lease timestamps. This is not merely an active client asking for a refund: all downstream test clients were gone well before the final observation. + +### Framework compatibility concern + +Installed framework source provides a specific lead: + +- Uvicorn's httptools protocol advertises ASGI HTTP 2.4. +- Its send function silently returns after downstream disconnection. +- Starlette's ASGI >=2.4 StreamingResponse path expects send to raise OSError for disconnect detection and does not run the older disconnect listener. + +This mismatch is consistent with streams continuing to consume upstream bytes while downstream sends become no-ops. Captured code is in the evidence directory. A runtime task-stack or controlled framework-version comparison is still needed for complete causal validation. + +Removing only LoggingMiddleware in a diagnostic router did not resolve disconnect renewal. Therefore, do not attribute the disconnect problem solely to that middleware. + +### Additional finalization/shutdown observation + +Forcibly stopping the dummy upstream finalized the diagnostic router's streams, but the unmodified router still had active reservations five seconds after upstream termination and required SIGKILL after a ten-second SIGTERM grace period. Its logs showed upstream termination warnings without completed settlement for those three requests in the captured window. The precise blocked operation was not traced. + +This adds a finalization/delivery investigation beyond transport inactivity. In this main reproduction, unlike the release reproduction, upstream termination did not promptly clear every reservation. + +### Updated conclusion + +The newer read timeout fixes silent upstream waits, but **does not eliminate reservation leaks for disconnected clients whose upstream streams keep producing bytes, or stalled downstream delivery**. The stream ownership/finalizer unit tests previously run do not exercise the complete real server/framework/middleware network path that exposed these cases. + +Prioritize real-network regression coverage and disconnect propagation, the installed server/framework compatibility, bounded downstream delivery and finalization, and an absolute request lifetime independent of keepalive traffic. No implementation fix has been made; alternate routes, multi-worker behavior, and database fault injection remain untested. diff --git a/reservation-repro-main/README.md b/reservation-repro-main/README.md new file mode 100644 index 00000000..e3df04b9 --- /dev/null +++ b/reservation-repro-main/README.md @@ -0,0 +1,106 @@ +# Current main: real-network streaming reservation reproductions + +## Tested version and environment + +- Commit: `96c8e2f77de8e9f8a0979d17dba0a6d20c78fe89` (local main at investigation time; no remote fetch was performed). +- Unpatched application built using its Dockerfile and frozen lockfile. +- Image: `localhost/routstr-reserved-repro:main`, ID `a787e603f565f3d34e1cc3999793d9dc2d2e3c968eb0ce0ded2f485450719bd0`. +- Podman 5.8.4; Python 3.14; Starlette 1.6.0; Uvicorn 0.31.1. +- Loopback ports 18090 (router), 18091 (dummy upstream), 18092 (diagnostic control). +- Separate container-local SQLite databases and synthetic balances; no original node data or secrets mounted. +- Read timeout accelerated to 3 seconds (confirmed effective); lease expiry to 6 seconds; heartbeat every 2 seconds. Background sweep remains 60 seconds. + +## Results + +| Scenario | Result | +| --- | --- | +| Finite stream with usage and DONE | Charged normally, zero reservation | +| One chunk then silence, client connected | Read timeout fired, estimated usage charged, zero reservation | +| One chunk then silence, client disconnected after 1 second | Reservation cleared on upstream read timeout; prompt disconnect cleanup was not demonstrated | +| No upstream response headers | Timeout produced HTTP 424; reservation released | +| Endless content stream, client disconnected after 1 second | Continued renewing; exact refund HTTP 400 persisted across background sweep | +| SSE comment-only keepalives every 0.5 seconds | No meaningful content or completion, but lease renewed and refund blocked; remained active after client disconnected | +| Flood stream to client that never reads | Lease renewed while client was stalled; still renewed after client socket closed | + +The three problematic streams retained 11-msat reservations through the full observation window. They began at timestamp 1790766366; at 1790766635, all remained active with lease timestamps 1790766634. Thus renewal continued for roughly 269 seconds, far beyond the 3-second read timeout, 6-second lease timeout, and multiple 60-second sweep intervals. All test clients were gone by approximately 1790766439. + +This proves persistence for minutes, not a measured days-long run. No new inference requests were made for the keys during observation; refund probes did not renew the leases. + +The flood scenario sends 64-KiB content deltas rapidly and uses a 1-KiB client receive buffer. It exercises a real non-reading downstream socket, but no live task-stack capture was collected to establish the precise blocked await at each snapshot. + +## Why the newer timeout is insufficient + +The read timeout is an inactivity timeout for upstream reads. Endless content or SSE keepalive bytes avoid it. A downstream-send wait is not bounded by it. + +More importantly, the runtime did not reliably propagate downstream disconnect into termination of these streams. Closed clients left upstream connections established and reservation owners alive, so heartbeats kept making the durable rows fresh. The sweeper therefore correctly declined to release them under its current policy. + +## Framework evidence and diagnostic control + +Captured sources (`starlette-source.txt`, `uvicorn-source.txt`) show: + +- Uvicorn 0.31.1's httptools protocol advertises ASGI HTTP spec 2.4. +- Its `send()` returns silently when `self.disconnected` is true; it does not raise an OSError. +- Starlette's StreamingResponse for ASGI >=2.4 relies on a send OSError to signal client disconnect, rather than running its older explicit disconnect listener. +- BaseHTTPMiddleware's outer streaming wrapper also does not explicitly listen for disconnect. + +This is a concrete framework compatibility concern consistent with the observations. Deterministic confirmation via a server-version/spec comparison or task instrumentation remains future work. + +A diagnostic second router removed only LoggingMiddleware using `no_logging_app.py`. Endless and keepalive clients still left active reservations after disconnect (`control-results.txt`). Thus LoggingMiddleware alone is not sufficient to explain the disconnect leak in this environment. This control is not a proposed production patch. + +When the dummy upstream was forcibly stopped, the control router finalized both streams. The unmodified main router still showed the three reservations active five seconds afterward and subsequently needed SIGKILL after a ten-second shutdown grace period. Logs showed upstream termination warnings but no completed settlement for those three in the captured window. The exact finalization blockage was not traced; it should be investigated separately, potentially including middleware delivery/backpressure interactions. Do not assert that upstream termination always clears these main reservations. + +## Reproduce + +From project root: + +```bash +podman build --build-arg GIT_COMMIT=$(git rev-parse HEAD) --build-arg GIT_TAG=main \ + -t localhost/routstr-reserved-repro:main . + +podman run -d --name reserved-dummy-main --network host \ + -v "$PWD/reservation-repro-main:/repro:ro,Z" \ + --entrypoint /.venv/bin/python localhost/routstr-reserved-repro:main \ + -m uvicorn dummy_upstream:app --app-dir /repro --host 127.0.0.1 --port 18091 + +podman run -d --name reserved-router-main --network host \ + -e DATABASE_URL=sqlite+aiosqlite:////tmp/reserved-main.db \ + -e UPSTREAM_BASE_URL=http://127.0.0.1:18091/v1 -e UPSTREAM_API_KEY=dummy \ + -e STALE_RESERVATION_TIMEOUT_SECONDS=6 -e UPSTREAM_READ_TIMEOUT=3 \ + -e CASHU_MINTS= -e ENABLE_PRICING_REFRESH=false \ + -e MODELS_REFRESH_INTERVAL_SECONDS=0 -e ADMIN_PASSWORD=local-repro-only \ + --entrypoint /.venv/bin/python localhost/routstr-reserved-repro:main \ + -m uvicorn routstr.core.main:app --host 127.0.0.1 --port 18090 +``` + +Wait for application startup and verify `/v1/models` includes gpt-4o-mini. Model/pricing discovery uses external services; this is not fully offline. + +```bash +podman exec -i reserved-router-main /.venv/bin/python - <<'PY' +import asyncio +from routstr.core.db import ApiKey, create_session +async def main(): + async with create_session() as s: + for k in ['finite','silent','silent-disconnect','endless-disconnect','keepalive','flood','header']: + s.add(ApiKey(hashed_key='main-'+k, balance=1000000000)) + await s.commit() +asyncio.run(main()) +PY + +.venv/bin/python reservation-repro-main/probe.py +``` + +The probe runs approximately 80 seconds, snapshots the DB, attempts refunds only on reserved keys (not actual Cashu payouts), and closes all clients. Later DB snapshots show continued renewal. Use fresh container names/databases on repeats or deliberately remove only the retained reproduction containers first. Do not overwrite original node containers. + +## Evidence and remaining work + +- `results.txt`: scenario matrix snapshots and refund errors. +- `connections.txt`: upstream sockets remained after downstream sockets disappeared. +- `final-before-stop.json`: continued renewal roughly 269 seconds after start. +- `after-upstream-stop.json`: reservations still active in unmodified main five seconds after upstream termination. +- `router.log`, `upstream.log`: application evidence before router shutdown. +- `control-results.txt`, `control-router.log`: comparison without LoggingMiddleware. +- `starlette-source.txt`, `uvicorn-source.txt`: installed framework behavior. + +Need: real-network regression tests, framework compatibility correction/verification, explicit disconnect monitoring that reaches upstream ownership, bounded downstream delivery, total request lifetime, and task-stack diagnostics for finalization stalls. Database fault injection, restart/multi-worker behavior, and alternate API routes were not tested here. + +All three main reproduction containers were stopped. The unmodified main router required SIGKILL; its retained database may contain active reservations. No application source fixes were made. diff --git a/reservation-repro-main/after-upstream-stop.json b/reservation-repro-main/after-upstream-stop.json new file mode 100644 index 00000000..a4c70696 --- /dev/null +++ b/reservation-repro-main/after-upstream-stop.json @@ -0,0 +1 @@ +{"time": 1790766644.5905168, "keys": [["main-finite", 0], ["main-silent", 0], ["main-silent-disconnect", 0], ["main-endless-disconnect", 11], ["main-keepalive", 11], ["main-flood", 11], ["main-header", 0]], "rows": [["main-flood", "active", 1790766636], ["main-keepalive", "active", 1790766636], ["main-endless-disconnect", "active", 1790766636], ["main-silent", "charged", 1790766368], ["main-silent-disconnect", "charged", 1790766368], ["main-header", "released", 1790766368], ["main-finite", "charged", 1790766366]]} diff --git a/reservation-repro-main/connections.txt b/reservation-repro-main/connections.txt new file mode 100644 index 00000000..2911cbec --- /dev/null +++ b/reservation-repro-main/connections.txt @@ -0,0 +1,7 @@ +ESTAB 0 0 127.0.0.1:18091 127.0.0.1:36380 users:(("python",pid=1480010,fd=7)) +ESTAB 0 0 127.0.0.1:36380 127.0.0.1:18091 users:(("python",pid=1480035,fd=31)) +ESTAB 0 0 127.0.0.1:36394 127.0.0.1:18091 users:(("python",pid=1480035,fd=32)) +ESTAB 0 0 127.0.0.1:36402 127.0.0.1:18091 users:(("python",pid=1480035,fd=33)) +CLOSE-WAIT 1 0 127.0.0.1:36456 127.0.0.1:18091 users:(("python",pid=1480035,fd=37)) +ESTAB 0 188 127.0.0.1:18091 127.0.0.1:36402 users:(("python",pid=1480010,fd=9)) +ESTAB 0 0 127.0.0.1:18091 127.0.0.1:36394 users:(("python",pid=1480010,fd=8)) diff --git a/reservation-repro-main/control-results.txt b/reservation-repro-main/control-results.txt new file mode 100644 index 00000000..84aee312 --- /dev/null +++ b/reservation-repro-main/control-results.txt @@ -0,0 +1,5 @@ +endless-control 200 +keepalive-control 200 +[('endless-control', 12), ('keepalive-control', 12)] +[('endless-control', 'active'), ('keepalive-control', 'active')] + diff --git a/reservation-repro-main/control-router.log b/reservation-repro-main/control-router.log new file mode 100644 index 00000000..de3a8147 --- /dev/null +++ b/reservation-repro-main/control-router.log @@ -0,0 +1,43 @@ +/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block + return result +2026-09-30 11:08:57 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). +2026-09-30 11:08:57 INFO uvicorn.error Started server process [1] +2026-09-30 11:08:57 INFO uvicorn.error Waiting for application startup. +2026-09-30 11:08:57 INFO routstr.core.main Application startup initiated +2026-09-30 11:08:59 INFO routstr.core.db Database migrations completed successfully +2026-09-30 11:08:59 INFO routstr.core.db Reset reserved balances on startup +2026-09-30 11:08:59 INFO routstr.upstream.helpers Seeding custom provider +2026-09-30 11:08:59 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings +2026-09-30 11:09:00 INFO routstr.proxy Initialized 1 upstream providers +2026-09-30 11:09:00 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider +2026-09-30 11:09:00 INFO routstr.nostr.analytics Usage analytics sharing task started +2026-09-30 11:09:00 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr +2026-09-30 11:09:00 INFO routstr.auth Dead-key pruning disabled (interval <= 0) +2026-09-30 11:09:00 INFO uvicorn.error Application startup complete. +2026-09-30 11:09:00 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18092 (Press CTRL+C to quit) +2026-09-30 11:09:30 INFO routstr.upstream.auto_topup Auto top-up worker started +2026-09-30 11:09:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:09:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:09:37 INFO routstr.auth Processing payment for request +2026-09-30 11:09:37 INFO routstr.auth Existing sk- API key found +2026-09-30 11:09:37 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:09:37 INFO routstr.auth Processing payment for request +2026-09-30 11:09:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:09:37 INFO routstr.payments RESERVE +2026-09-30 11:09:37 INFO routstr.auth Payment processed successfully +2026-09-30 11:09:37 INFO routstr.payments RESERVE +2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete +2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete +2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:10:38 INFO routstr.auth Calculated token-based cost +2026-09-30 11:10:38 INFO routstr.auth Finalized payment with additional charge +2026-09-30 11:10:38 INFO routstr.payments FINALIZE +2026-09-30 11:10:38 INFO routstr.auth Payment settlement finished +2026-09-30 11:10:38 INFO routstr.auth Calculated token-based cost +2026-09-30 11:10:38 INFO routstr.auth Refunding excess payment +2026-09-30 11:10:38 INFO routstr.auth Refund processed successfully +2026-09-30 11:10:38 INFO routstr.payments FINALIZE +2026-09-30 11:10:38 INFO routstr.auth Payment settlement finished diff --git a/reservation-repro-main/dummy_upstream.py b/reservation-repro-main/dummy_upstream.py new file mode 100644 index 00000000..1be75ecc --- /dev/null +++ b/reservation-repro-main/dummy_upstream.py @@ -0,0 +1,45 @@ +"""Loopback-only streaming fixture; no router monkeypatches.""" +import asyncio +import json +import time +from fastapi import FastAPI, Request +from fastapi.responses import StreamingResponse + +app = FastAPI() +events = [] + +@app.get('/events') +async def history(): + return events + +@app.get('/v1/models') +async def models(): + return {'object': 'list', 'data': [{'id': 'gpt-4o-mini', 'object': 'model', 'created': 1, 'owned_by': 'repro'}]} + +@app.post('/v1/chat/completions') +async def completions(request: Request): + body = await request.json() + mode = body.get('messages', [{}])[0].get('content', 'finite') + events.append({'event': 'start', 'mode': mode, 'time': time.time()}) + if mode.startswith('header'): + await asyncio.sleep(3600) + async def stream(): + count = 0 + try: + while True: + if mode.startswith('keepalive'): + yield ': ping\n\n' + else: + chunk = {'id': 'repro', 'object': 'chat.completion.chunk', 'created': int(time.time()), 'model': 'gpt-4o-mini', 'choices': [{'index': 0, 'delta': {'content': 'x' * (65536 if mode.startswith('flood') else 1)}, 'finish_reason': None}]} + yield 'data: ' + json.dumps(chunk) + '\n\n' + count += 1 + if mode == 'finite' and count >= 3: + yield 'data: ' + json.dumps({'id': 'repro', 'object': 'chat.completion.chunk', 'model': 'gpt-4o-mini', 'choices': [], 'usage': {'prompt_tokens': 1, 'completion_tokens': count, 'total_tokens': count + 1}}) + '\n\n' + yield 'data: [DONE]\n\n' + return + await asyncio.sleep(3600 if mode.startswith('silent') else (0.001 if mode.startswith('flood') else 0.5)) + finally: + event = {'event': 'close', 'mode': mode, 'chunks': count, 'time': time.time()} + events.append(event) + print(json.dumps(event), flush=True) + return StreamingResponse(stream(), media_type='text/event-stream') diff --git a/reservation-repro-main/final-before-stop.json b/reservation-repro-main/final-before-stop.json new file mode 100644 index 00000000..4647ae85 --- /dev/null +++ b/reservation-repro-main/final-before-stop.json @@ -0,0 +1 @@ +{"time": 1790766635.1679196, "keys": [["main-finite", 0], ["main-silent", 0], ["main-silent-disconnect", 0], ["main-endless-disconnect", 11], ["main-keepalive", 11], ["main-flood", 11], ["main-header", 0]], "rows": [["main-flood", "active", 1790766634], ["main-keepalive", "active", 1790766634], ["main-endless-disconnect", "active", 1790766634], ["main-silent", "charged", 1790766368], ["main-silent-disconnect", "charged", 1790766368], ["main-header", "released", 1790766368], ["main-finite", "charged", 1790766366]]} diff --git a/reservation-repro-main/no_logging_app.py b/reservation-repro-main/no_logging_app.py new file mode 100644 index 00000000..16c4cdfc --- /dev/null +++ b/reservation-repro-main/no_logging_app.py @@ -0,0 +1,4 @@ +"""Diagnostic comparison ONLY: remove LoggingMiddleware from unchanged image app.""" +from routstr.core.main import app +from routstr.core.middleware import LoggingMiddleware +app.user_middleware = [m for m in app.user_middleware if m.cls is not LoggingMiddleware] diff --git a/reservation-repro-main/probe.py b/reservation-repro-main/probe.py new file mode 100644 index 00000000..57653e5e --- /dev/null +++ b/reservation-repro-main/probe.py @@ -0,0 +1,60 @@ +import asyncio +import json +import socket +import subprocess +import time +import httpx + +BASE='http://127.0.0.1:18090' + +def snapshot(): + code="import sqlite3,json,time; c=sqlite3.connect('/tmp/reserved-main.db'); c.row_factory=sqlite3.Row; print(json.dumps({'time':time.time(),'keys':[dict(r) for r in c.execute(\"select hashed_key,balance,reserved_balance,reserved_at from api_keys where hashed_key like 'main-%'\")],'rows':[dict(r) for r in c.execute(\"select * from reservation_releases where key_hash like 'main-%'\")]}))" + return json.loads(subprocess.check_output(['podman','exec','reserved-router-main','/.venv/bin/python','-c',code],text=True)) + +async def consume(mode): + try: + async with httpx.AsyncClient(timeout=None) as c: + async with c.stream('POST',BASE+'/v1/chat/completions',headers={'Authorization':'Bearer sk-main-'+mode},json={'model':'gpt-4o-mini','messages':[{'role':'user','content':mode}],'stream':True,'max_tokens':10}) as r: + print('STREAM',mode,r.status_code,flush=True) + async for _ in r.aiter_bytes(): pass + print('ENDED',mode,flush=True) + except asyncio.CancelledError: + print('CLIENT_DISCONNECTED',mode,flush=True) + raise + except Exception as e: + print('CLIENT_ERROR',mode,type(e).__name__,str(e),flush=True) + +async def report(label): + print(label,json.dumps(snapshot()),flush=True) + async with httpx.AsyncClient(timeout=5) as c: + for mode in ['silent-disconnect','endless-disconnect','keepalive','flood','header']: + # Only attempt payout while reserved: avoid requiring a real mint. + if next(k for k in snapshot()['keys'] if k['hashed_key']=='main-'+mode)['reserved_balance']: + r=await c.post(BASE+'/v1/wallet/refund',headers={'Authorization':'Bearer sk-main-'+mode}) + print('REFUND',mode,r.status_code,r.text,flush=True) + print('UPSTREAM_EVENTS',json.dumps((await c.get('http://127.0.0.1:18091/events')).json()),flush=True) + +async def main(): + modes=['finite','silent','silent-disconnect','endless-disconnect','keepalive','header'] + tasks={m:asyncio.create_task(consume(m)) for m in modes} + # Real client with a small receive buffer, never draining the HTTP response. + sock=socket.socket(); sock.setsockopt(socket.SOL_SOCKET,socket.SO_RCVBUF,1024); sock.connect(('127.0.0.1',18090)) + body=json.dumps({'model':'gpt-4o-mini','messages':[{'role':'user','content':'flood'}],'stream':True,'max_tokens':10}).encode() + sock.sendall(b'POST /v1/chat/completions HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer sk-main-flood\r\nContent-Type: application/json\r\nContent-Length: '+str(len(body)).encode()+b'\r\n\r\n'+body) + await asyncio.sleep(1) + for m in ['silent-disconnect','endless-disconnect']: + tasks[m].cancel() + await asyncio.gather(tasks['silent-disconnect'],tasks['endless-disconnect'],return_exceptions=True) + await asyncio.sleep(9) + await report('AT_10_SECONDS') + await asyncio.sleep(60) + await report('AFTER_SWEEP') + sock.close() + tasks['keepalive'].cancel() + await asyncio.gather(tasks['keepalive'],return_exceptions=True) + await asyncio.sleep(8) + await report('AFTER_ALL_CLIENTS_CLOSED') + for task in tasks.values(): task.cancel() + await asyncio.gather(*tasks.values(),return_exceptions=True) + +asyncio.run(main()) diff --git a/reservation-repro-main/results.txt b/reservation-repro-main/results.txt new file mode 100644 index 00000000..7f21d733 --- /dev/null +++ b/reservation-repro-main/results.txt @@ -0,0 +1,27 @@ +STREAM keepalive 200 +STREAM endless-disconnect 200 +STREAM silent 200 +STREAM silent-disconnect 200 +STREAM finite 200 +CLIENT_DISCONNECTED silent-disconnect +CLIENT_DISCONNECTED endless-disconnect +ENDED finite +ENDED silent +STREAM header 424 +ENDED header +AT_10_SECONDS {"time": 1790766376.4572322, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "50569f573caf4d6fb7916da3570493db", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 11, "status": "active", "created_at": 1790766376}, {"id": "0ffb4d61dc0d4aaf9e518c74b7afd1bb", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 11, "status": "active", "created_at": 1790766376}, {"id": "66d5bdd9d9814f6fb1576ed6708f431e", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 11, "status": "active", "created_at": 1790766374}, {"id": "a891d80b8db64e488f8896936cd5f2fe", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "6ebb7f0829bd4f569d9d2516dabd9fed", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "e14358bc6e7249c0ac7335c9d87e7b43", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 11, "status": "released", "created_at": 1790766368}, {"id": "26c897009e294f94b75ca51f071d335a", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 11, "status": "charged", "created_at": 1790766366}]} +REFUND endless-disconnect 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"c272c298-bace-482c-926f-0c56fdaeaa5e"} +REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"3c1124f6-1c21-4417-b4fb-2ffdead58c31"} +REFUND flood 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"afca3cab-361a-48f6-85e9-58247c17a5f2"} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790766366.393702}, {"event": "start", "mode": "keepalive", "time": 1790766366.408879}, {"event": "start", "mode": "endless-disconnect", "time": 1790766366.4305305}, {"event": "start", "mode": "silent", "time": 1790766366.4564564}, {"event": "start", "mode": "silent-disconnect", "time": 1790766366.4789124}, {"event": "start", "mode": "header", "time": 1790766366.5032742}, {"event": "start", "mode": "finite", "time": 1790766366.5216281}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321}] +AFTER_SWEEP {"time": 1790766437.9892845, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "50569f573caf4d6fb7916da3570493db", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 11, "status": "active", "created_at": 1790766436}, {"id": "0ffb4d61dc0d4aaf9e518c74b7afd1bb", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 11, "status": "active", "created_at": 1790766436}, {"id": "66d5bdd9d9814f6fb1576ed6708f431e", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 11, "status": "active", "created_at": 1790766436}, {"id": "a891d80b8db64e488f8896936cd5f2fe", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "6ebb7f0829bd4f569d9d2516dabd9fed", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "e14358bc6e7249c0ac7335c9d87e7b43", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 11, "status": "released", "created_at": 1790766368}, {"id": "26c897009e294f94b75ca51f071d335a", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 11, "status": "charged", "created_at": 1790766366}]} +REFUND endless-disconnect 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"4a00cdce-d049-4be8-940f-2a652349c1f9"} +REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"bcf3b7d4-ae59-494c-8049-bb43476211e8"} +REFUND flood 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"4b459580-84d1-480a-b1cb-28898a04395e"} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790766366.393702}, {"event": "start", "mode": "keepalive", "time": 1790766366.408879}, {"event": "start", "mode": "endless-disconnect", "time": 1790766366.4305305}, {"event": "start", "mode": "silent", "time": 1790766366.4564564}, {"event": "start", "mode": "silent-disconnect", "time": 1790766366.4789124}, {"event": "start", "mode": "header", "time": 1790766366.5032742}, {"event": "start", "mode": "finite", "time": 1790766366.5216281}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321}] +CLIENT_DISCONNECTED keepalive +AFTER_ALL_CLIENTS_CLOSED {"time": 1790766447.4688976, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "50569f573caf4d6fb7916da3570493db", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 11, "status": "active", "created_at": 1790766447}, {"id": "0ffb4d61dc0d4aaf9e518c74b7afd1bb", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 11, "status": "active", "created_at": 1790766446}, {"id": "66d5bdd9d9814f6fb1576ed6708f431e", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 11, "status": "active", "created_at": 1790766446}, {"id": "a891d80b8db64e488f8896936cd5f2fe", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "6ebb7f0829bd4f569d9d2516dabd9fed", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "e14358bc6e7249c0ac7335c9d87e7b43", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 11, "status": "released", "created_at": 1790766368}, {"id": "26c897009e294f94b75ca51f071d335a", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 11, "status": "charged", "created_at": 1790766366}]} +REFUND endless-disconnect 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"67367fbb-2fd6-4ce7-a0e3-3b1d0f30cb76"} +REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"96f8b70c-584e-476d-aa9c-ea9092e9393f"} +REFUND flood 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"60ac9762-dc92-4e08-9456-77346a70a63d"} +UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790766366.393702}, {"event": "start", "mode": "keepalive", "time": 1790766366.408879}, {"event": "start", "mode": "endless-disconnect", "time": 1790766366.4305305}, {"event": "start", "mode": "silent", "time": 1790766366.4564564}, {"event": "start", "mode": "silent-disconnect", "time": 1790766366.4789124}, {"event": "start", "mode": "header", "time": 1790766366.5032742}, {"event": "start", "mode": "finite", "time": 1790766366.5216281}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321}] diff --git a/reservation-repro-main/router.log b/reservation-repro-main/router.log new file mode 100644 index 00000000..f2e7c738 --- /dev/null +++ b/reservation-repro-main/router.log @@ -0,0 +1,114 @@ +/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block + return result +2026-09-30 11:05:28 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). +2026-09-30 11:05:28 INFO uvicorn.error Started server process [1] +2026-09-30 11:05:28 INFO uvicorn.error Waiting for application startup. +2026-09-30 11:05:28 INFO routstr.core.main Application startup initiated +2026-09-30 11:05:30 INFO routstr.core.db Database migrations completed successfully +2026-09-30 11:05:30 INFO routstr.core.db Reset reserved balances on startup +2026-09-30 11:05:30 INFO routstr.upstream.helpers Seeding custom provider +2026-09-30 11:05:30 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings +2026-09-30 11:05:31 INFO routstr.proxy Initialized 1 upstream providers +2026-09-30 11:05:31 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider +2026-09-30 11:05:31 INFO routstr.nostr.analytics Usage analytics sharing task started +2026-09-30 11:05:31 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr +2026-09-30 11:05:31 INFO routstr.auth Dead-key pruning disabled (interval <= 0) +2026-09-30 11:05:31 INFO uvicorn.error Application startup complete. +2026-09-30 11:05:31 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18090 (Press CTRL+C to quit) +2026-09-30 11:06:01 INFO routstr.upstream.auto_topup Auto top-up worker started +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found +2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully +2026-09-30 11:06:06 INFO routstr.auth Processing payment for request +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully +2026-09-30 11:06:06 INFO routstr.payments RESERVE +2026-09-30 11:06:07 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:06:07 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:06:07 INFO routstr.auth Calculated token-based cost +2026-09-30 11:06:07 INFO routstr.auth Refunding excess payment +2026-09-30 11:06:07 INFO routstr.auth Refund processed successfully +2026-09-30 11:06:07 INFO routstr.payments FINALIZE +2026-09-30 11:06:07 INFO routstr.auth Payment settlement finished +2026-09-30 11:06:09 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream +2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:06:09 INFO routstr.auth Calculated token-based cost +2026-09-30 11:06:09 INFO routstr.auth Refunding excess payment +2026-09-30 11:06:09 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream +2026-09-30 11:06:09 INFO routstr.auth Refund processed successfully +2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Applied model-specific pricing +2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Calculated token-based cost +2026-09-30 11:06:09 INFO routstr.auth Calculated token-based cost +2026-09-30 11:06:09 INFO routstr.auth Refunding excess payment +2026-09-30 11:06:09 ERROR routstr.upstream.base HTTP request error to upstream +2026-09-30 11:06:09 WARNING routstr.proxy Upstream base failed for model=gpt-4o-mini: Upstream service request timed out +2026-09-30 11:06:09 INFO routstr.auth Refund processed successfully +2026-09-30 11:06:09 INFO routstr.payments FINALIZE +2026-09-30 11:06:09 INFO routstr.auth Payment settlement finished +2026-09-30 11:06:09 ERROR routstr.core.exceptions Unhandled exception +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:06:09 ERROR uvicorn.error Exception in ASGI application +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:06:09 INFO routstr.payments FINALIZE +2026-09-30 11:06:09 INFO routstr.auth Payment settlement finished +2026-09-30 11:06:09 ERROR routstr.core.exceptions Unhandled exception +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:06:09 ERROR uvicorn.error Exception in ASGI application +httpcore.ReadTimeout + +The above exception was the direct cause of the following exception: + +httpx.ReadTimeout +2026-09-30 11:06:16 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:06:17 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:06:17 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:07:18 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:07:18 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:07:19 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:07:28 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:07:28 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:07:28 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. +2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete +2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete +2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete diff --git a/reservation-repro-main/starlette-source.txt b/reservation-repro-main/starlette-source.txt new file mode 100644 index 00000000..78736b3f --- /dev/null +++ b/reservation-repro-main/starlette-source.txt @@ -0,0 +1,169 @@ + async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: + if scope["type"] != "http": + await self.app(scope, receive, send) + return + + request = _CachedRequest(scope, receive) + wrapped_receive = request.wrapped_receive + response_sent = anyio.Event() + app_exc: Exception | None = None + exception_already_raised = False + + async def call_next(request: Request) -> Response: + async def receive_or_disconnect() -> Message: + if response_sent.is_set(): + return {"type": "http.disconnect"} + + async with anyio.create_task_group() as task_group: + + async def wrap(func: Callable[[], Awaitable[T]]) -> T: + result = await func() + task_group.cancel_scope.cancel() + return result + + task_group.start_soon(wrap, response_sent.wait) + message = await wrap(wrapped_receive) + + if response_sent.is_set(): + return {"type": "http.disconnect"} + + return message + + async def send_no_error(message: Message) -> None: + try: + await send_stream.send(message) + except anyio.BrokenResourceError: + # recv_stream has been closed, i.e. response_sent has been set. + return + + async def coro() -> None: + nonlocal app_exc + + with send_stream: + try: + await self.app(scope, receive_or_disconnect, send_no_error) + except Exception as exc: + app_exc = exc + + task_group.start_soon(coro) + + try: + message = await recv_stream.receive() + info = message.get("info", None) + if message["type"] == "http.response.debug" and info is not None: + message = await recv_stream.receive() + except anyio.EndOfStream: + if app_exc is not None: + nonlocal exception_already_raised + exception_already_raised = True + # Prevent `anyio.EndOfStream` from polluting app exception context. + # If both cause and context are None then the context is suppressed + # and `anyio.EndOfStream` is not present in the exception traceback. + # If exception cause is not None then it is propagated with + # reraising here. + # If exception has no cause but has context set then the context is + # propagated as a cause with the reraise. This is necessary in order + # to prevent `anyio.EndOfStream` from polluting the exception + # context. + raise app_exc from app_exc.__cause__ or app_exc.__context__ + raise RuntimeError("No response returned.") + + assert message["type"] == "http.response.start" + + async def body_stream() -> BodyStreamGenerator: + async for message in recv_stream: + if message["type"] == "http.response.pathsend": + yield message + break + assert message["type"] == "http.response.body", f"Unexpected message: {message}" + body = message.get("body", b"") + if body: + yield body + if not message.get("more_body", False): + break + + response = _StreamingResponse(status_code=message["status"], content=body_stream(), info=info) + response.raw_headers = message["headers"] + return response + + streams: anyio.create_memory_object_stream[Message] = anyio.create_memory_object_stream() + send_stream, recv_stream = streams + with recv_stream, send_stream: + async with create_collapsing_task_group() as task_group: + response = await self.dispatch_func(request, call_next) + await response(scope, wrapped_receive, send) + response_sent.set() + recv_stream.close() + if app_exc is not None and not exception_already_raised: + raise app_exc + +class _StreamingResponse(Response): + def __init__( + self, + content: AsyncContentStream, + status_code: int = 200, + headers: Mapping[str, str] | None = None, + media_type: str | None = None, + info: Mapping[str, Any] | None = None, + ) -> None: + self.info = info + self.body_iterator = content + self.status_code = status_code + self.media_type = media_type + self.init_headers(headers) + self.background = None + + async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: + if self.info is not None: + await send({"type": "http.response.debug", "info": self.info}) + await send( + { + "type": "http.response.start", + "status": self.status_code, + "headers": self.raw_headers, + } + ) + + should_close_body = True + async for chunk in self.body_iterator: + if isinstance(chunk, dict): + # We got an ASGI message which is not response body (eg: pathsend) + should_close_body = False + await send(chunk) + continue + await send({"type": "http.response.body", "body": chunk, "more_body": True}) + + if should_close_body: + await send({"type": "http.response.body", "body": b"", "more_body": False}) + + if self.background: + await self.background() + + async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: + if scope["type"] == "websocket": + send = self._wrap_websocket_denial_send(send) + await self.stream_response(send) + if self.background is not None: + await self.background() + return + + spec_version = tuple(map(int, scope.get("asgi", {}).get("spec_version", "2.0").split("."))) + + if spec_version >= (2, 4): + try: + await self.stream_response(send) + except OSError: + raise ClientDisconnect() + else: + async with create_collapsing_task_group() as task_group: + + async def wrap(func: Callable[[], Awaitable[None]]) -> None: + await func() + task_group.cancel_scope.cancel() + + task_group.start_soon(wrap, partial(self.stream_response, send)) + await wrap(partial(self.listen_for_disconnect, receive)) + + if self.background is not None: + await self.background() + diff --git a/reservation-repro-main/upstream.log b/reservation-repro-main/upstream.log new file mode 100644 index 00000000..2f124bcd --- /dev/null +++ b/reservation-repro-main/upstream.log @@ -0,0 +1,23 @@ +/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block + return result +INFO: Started server process [1] +INFO: Waiting for application startup. +INFO: Application startup complete. +INFO: Uvicorn running on http://127.0.0.1:18091 (Press CTRL+C to quit) +INFO: 127.0.0.1:59686 - "GET /v1/models HTTP/1.1" 200 OK +INFO: 127.0.0.1:36380 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:36394 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:36402 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:36418 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:36434 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:36456 - "POST /v1/chat/completions HTTP/1.1" 200 OK +{"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321} +INFO: 127.0.0.1:54322 - "GET /events HTTP/1.1" 200 OK +INFO: 127.0.0.1:51770 - "GET /events HTTP/1.1" 200 OK +INFO: 127.0.0.1:42140 - "GET /events HTTP/1.1" 200 OK +INFO: 127.0.0.1:50608 - "GET /v1/models HTTP/1.1" 200 OK +INFO: 127.0.0.1:46164 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:46178 - "POST /v1/chat/completions HTTP/1.1" 200 OK +INFO: 127.0.0.1:39728 - "GET /events HTTP/1.1" 200 OK +INFO: Shutting down +INFO: Waiting for connections to close. (CTRL+C to force quit) diff --git a/reservation-repro-main/uvicorn-source.txt b/reservation-repro-main/uvicorn-source.txt new file mode 100644 index 00000000..2c5d74b3 --- /dev/null +++ b/reservation-repro-main/uvicorn-source.txt @@ -0,0 +1,125 @@ + async def send(self, message: ASGISendEvent) -> None: + message_type = message["type"] + + if self.flow.write_paused and not self.disconnected: + await self.flow.drain() # pragma: full coverage + + if self.disconnected: + return # pragma: full coverage + + if not self.response_started: + # Sending response status line and headers + if message_type != "http.response.start": + msg = "Expected ASGI message 'http.response.start', but got '%s'." + raise RuntimeError(msg % message_type) + message = cast("HTTPResponseStartEvent", message) + + self.response_started = True + self.waiting_for_100_continue = False + + status_code = message["status"] + headers = self.default_headers + list(message.get("headers", [])) + + if CLOSE_HEADER in self.scope["headers"] and CLOSE_HEADER not in headers: + headers = headers + [CLOSE_HEADER] + + if self.access_log: + self.access_logger.info( + '%s - "%s %s HTTP/%s" %d', + get_client_addr(self.scope), + self.scope["method"], + get_path_with_query_string(self.scope), + self.scope["http_version"], + status_code, + ) + + # Write response status line and headers + content = [STATUS_LINE[status_code]] + + for name, value in headers: + if HEADER_RE.search(name): + raise RuntimeError("Invalid HTTP header name.") # pragma: full coverage + if HEADER_VALUE_RE.search(value): + raise RuntimeError("Invalid HTTP header value.") + + name = name.lower() + if name == b"content-length" and self.chunked_encoding is None: + self.expected_content_length = int(value.decode()) + self.chunked_encoding = False + elif name == b"transfer-encoding" and value.lower() == b"chunked": + self.expected_content_length = 0 + self.chunked_encoding = True + elif name == b"connection" and value.lower() == b"close": + self.keep_alive = False + content.extend([name, b": ", value, b"\r\n"]) + + if self.chunked_encoding is None and self.scope["method"] != "HEAD" and status_code not in (204, 304): + # Neither content-length nor transfer-encoding specified + self.chunked_encoding = True + content.append(b"transfer-encoding: chunked\r\n") + + content.append(b"\r\n") + self.transport.write(b"".join(content)) + + elif not self.response_complete: + # Sending response body + if message_type != "http.response.body": + msg = "Expected ASGI message 'http.response.body', but got '%s'." + raise RuntimeError(msg % message_type) + + body = cast(bytes, message.get("body", b"")) + more_body = message.get("more_body", False) + + # Write response body + if self.scope["method"] == "HEAD": + self.expected_content_length = 0 + elif self.chunked_encoding: + if body: + content = [b"%x\r\n" % len(body), body, b"\r\n"] + else: + content = [] + if not more_body: + content.append(b"0\r\n\r\n") + self.transport.write(b"".join(content)) + else: + num_bytes = len(body) + if num_bytes > self.expected_content_length: + raise RuntimeError("Response content longer than Content-Length") + else: + self.expected_content_length -= num_bytes + self.transport.write(body) + + # Handle response completion + if not more_body: + if self.expected_content_length != 0: + raise RuntimeError("Response content shorter than Content-Length") + self.response_complete = True + self.message_event.set() + if not self.keep_alive: + self.transport.close() + self.on_response() + + else: + # Response already sent + msg = "Unexpected ASGI message '%s' sent, after response already completed." + raise RuntimeError(msg % message_type) + + def connection_lost(self, exc: Exception | None) -> None: + self.connections.discard(self) + + if self.logger.level <= TRACE_LOG_LEVEL: + prefix = "%s:%d - " % self.client if self.client else "" + self.logger.log(TRACE_LOG_LEVEL, "%sHTTP connection lost", prefix) + + if self.cycle and not self.cycle.response_complete: + self.cycle.disconnected = True + if self.cycle is not None: + self.cycle.message_event.set() + if self.flow is not None: + self.flow.resume_writing() + if exc is None: + self.transport.close() + self._unset_keepalive_if_required() + + self.parser = None + From 368ce4247e9cc2751ddceb178d1feb55cd11f068 Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Wed, 30 Sep 2026 20:29:18 +0800 Subject: [PATCH 10/12] fix: satisfy mypy for ASGI app awaitable in lifecycle middleware ASGIApp.__call__ is typed as returning Awaitable[None], not a Coroutine, so asyncio.create_task rejected it. asyncio.ensure_future accepts any awaitable and returns a Future, which supports every operation the middleware uses (cancel, done, exception, asyncio.wait, await). --- routstr/core/lifecycle.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/routstr/core/lifecycle.py b/routstr/core/lifecycle.py index 897df5b7..dad968ba 100644 --- a/routstr/core/lifecycle.py +++ b/routstr/core/lifecycle.py @@ -82,7 +82,9 @@ class RequestLifecycleMiddleware: response_started = True receiver = asyncio.create_task(pump()) - work = asyncio.create_task(self.app(scope, downstream_receive, downstream_send)) + work: asyncio.Future[None] = asyncio.ensure_future( + self.app(scope, downstream_receive, downstream_send) + ) gone = asyncio.create_task(disconnected.wait()) try: done, _ = await asyncio.wait( From 2b24342a7242acef9849e77e107d5e1ff8c51ed8 Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Wed, 30 Sep 2026 21:56:21 +0800 Subject: [PATCH 11/12] ci: exclude temporary repro artifacts from ruff/mypy, type new lifecycle test The investigation artifacts under repro/ and reservation-repro-main/ are deliberately kept for review but fail ruff and mypy, and the two directories both define dummy_upstream.py, which makes 'mypy .' abort with a duplicate-module error. Exclude them (mirroring examples/) via ruff extend-exclude and a mypy exclude, and fix the newly-surfaced mypy errors in tests/unit/test_request_lifecycle.py by annotating it. The previous [tool.ruff.lint] exclude was not applied; move it to [tool.ruff] extend-exclude so defaults are preserved. --- pyproject.toml | 5 ++++- tests/unit/test_request_lifecycle.py | 13 +++++++------ 2 files changed, 11 insertions(+), 7 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 296fab3f..1faf6828 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -72,10 +72,12 @@ build-backend = "setuptools.build_meta" [tool.setuptools] packages = ["routstr"] +[tool.ruff] +extend-exclude = ["examples", "repro", "reservation-repro-main"] + [tool.ruff.lint] select = ["E", "F", "I"] ignore = ["E501"] -exclude = ["examples"] [tool.mypy] python_version = "3.11" @@ -85,6 +87,7 @@ check_untyped_defs = true disallow_untyped_calls = true disallow_incomplete_defs = true disallow_untyped_decorators = true +exclude = ["^repro/", "^reservation-repro-main/"] [tool.uv.sources] routstr = { workspace = true } diff --git a/tests/unit/test_request_lifecycle.py b/tests/unit/test_request_lifecycle.py index 03cf9897..41c76e50 100644 --- a/tests/unit/test_request_lifecycle.py +++ b/tests/unit/test_request_lifecycle.py @@ -2,6 +2,7 @@ import asyncio from unittest.mock import patch import pytest +from starlette.types import Message, Receive, Scope, Send from routstr.core.lifecycle import RequestLifecycleMiddleware from routstr.core.settings import settings @@ -9,13 +10,13 @@ from routstr.core.settings import settings @pytest.mark.asyncio @pytest.mark.parametrize("reason", ["disconnect", "deadline", "send"]) -async def test_lifecycle_stops_live_work(reason): +async def test_lifecycle_stops_live_work(reason: str) -> None: closed = asyncio.Event() - receive_queue = asyncio.Queue() + receive_queue: asyncio.Queue[Message] = asyncio.Queue() await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) - sent = [] + sent: list[Message] = [] - async def app(scope, receive, send): + async def app(scope: Scope, receive: Receive, send: Send) -> None: try: assert (await receive())["type"] == "http.request" await send({"type": "http.response.start", "status": 200, "headers": []}) @@ -27,12 +28,12 @@ async def test_lifecycle_stops_live_work(reason): finally: closed.set() - async def send(message): + async def send(message: Message) -> None: sent.append(message) if reason == "send" and message["type"] == "http.response.body": await asyncio.sleep(100) - async def disconnect(): + async def disconnect() -> None: await asyncio.sleep(0.02) await receive_queue.put({"type": "http.disconnect"}) From 53a4f3216ca25981fbdb82ca91c9a961da5be03f Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 30 Sep 2026 21:47:47 +0200 Subject: [PATCH 12/12] fix: let stream finalizers settle billing and remove repro artifacts --- .env.example | 6 + RESERVED_BALANCE.md | 541 ------------------ pyproject.toml | 3 +- repro/IMPLEMENTATION.md | 46 -- repro/dummy_upstream.py | 45 -- repro/probe.py | 60 -- repro/results-final.txt | 19 - repro/results.txt | 19 - repro/router-final.log | 108 ---- repro/router-first.log | 115 ---- reservation-repro-main/README.md | 106 ---- .../after-upstream-stop.json | 1 - reservation-repro-main/connections.txt | 7 - reservation-repro-main/control-results.txt | 5 - reservation-repro-main/control-router.log | 43 -- reservation-repro-main/dummy_upstream.py | 45 -- reservation-repro-main/final-before-stop.json | 1 - reservation-repro-main/no_logging_app.py | 4 - reservation-repro-main/probe.py | 60 -- reservation-repro-main/results.txt | 27 - reservation-repro-main/router.log | 114 ---- reservation-repro-main/starlette-source.txt | 169 ------ reservation-repro-main/upstream.log | 23 - reservation-repro-main/uvicorn-source.txt | 125 ---- routstr/auth.py | 11 +- routstr/core/lifecycle.py | 63 +- tests/unit/test_request_lifecycle.py | 227 ++++++++ tests/unit/test_stale_reservations.py | 30 +- 28 files changed, 298 insertions(+), 1725 deletions(-) delete mode 100644 RESERVED_BALANCE.md delete mode 100644 repro/IMPLEMENTATION.md delete mode 100644 repro/dummy_upstream.py delete mode 100644 repro/probe.py delete mode 100644 repro/results-final.txt delete mode 100644 repro/results.txt delete mode 100644 repro/router-final.log delete mode 100644 repro/router-first.log delete mode 100644 reservation-repro-main/README.md delete mode 100644 reservation-repro-main/after-upstream-stop.json delete mode 100644 reservation-repro-main/connections.txt delete mode 100644 reservation-repro-main/control-results.txt delete mode 100644 reservation-repro-main/control-router.log delete mode 100644 reservation-repro-main/dummy_upstream.py delete mode 100644 reservation-repro-main/final-before-stop.json delete mode 100644 reservation-repro-main/no_logging_app.py delete mode 100644 reservation-repro-main/probe.py delete mode 100644 reservation-repro-main/results.txt delete mode 100644 reservation-repro-main/router.log delete mode 100644 reservation-repro-main/starlette-source.txt delete mode 100644 reservation-repro-main/upstream.log delete mode 100644 reservation-repro-main/uvicorn-source.txt diff --git a/.env.example b/.env.example index 5688f6c3..e79b7770 100644 --- a/.env.example +++ b/.env.example @@ -72,6 +72,12 @@ ROUTSTR_SECRET_KEY= # UPSTREAM_POOL_TIMEOUT=5 # UPSTREAM_READ_TIMEOUT=900 +# Request and reservation lifetime limits (seconds) +# STALE_RESERVATION_TIMEOUT_SECONDS=300 +# MAX_REQUEST_LIFETIME_SECONDS=1800 +# DOWNSTREAM_SEND_TIMEOUT_SECONDS=60 +# REQUEST_CLEANUP_TIMEOUT_SECONDS=30 + # Logging # LOG_LEVEL=INFO # ENABLE_CONSOLE_LOGGING=true diff --git a/RESERVED_BALANCE.md b/RESERVED_BALANCE.md deleted file mode 100644 index a7417bc4..00000000 --- a/RESERVED_BALANCE.md +++ /dev/null @@ -1,541 +0,0 @@ -# Reserved balance blocks refunds long after the last request - -## Reported issue - -A client attempting to refund an API key receives: - -> Cannot refund key. There are ongoing requests for this api key. - -The user reports that the key has not been used in a very long time, potentially days. This is not a refund racing with normal request completion. The expected behavior is that reservations left by disconnected, crashed, abandoned, or failed requests eventually expire and the key becomes refundable. - -The error does **not** prove that an upstream inference request is running. In the current implementation, it means the refund endpoint still sees a positive aggregate `reserved_balance` after attempting stale-reservation cleanup. - -This document records a source-code investigation of the current checkout. The affected node's database, logs, runtime tasks, effective configuration, and deployed version have not been inspected. The production root cause remains unconfirmed. - -## Investigation scope and results - -Checkout inspected: `96c8e2f7` (`Merge pull request #790 from Routstr/fix/rename-unsupported-param`). - -The existing cleanup system is implemented and wired into application startup. It protects several important accounting invariants, but it is based on renewable reservation leases rather than a hard maximum request lifetime. - -Verification command: - -```bash -.venv/bin/pytest \ - tests/unit/test_stale_reservations.py \ - tests/unit/test_streaming_billing_finalization.py \ - tests/integration/test_negative_available_balance_repro.py -q -``` - -Result: **59 passed in 10.04 seconds**. - -These passing tests verify existing recovery paths; they do not establish what happened on the affected node or demonstrate recovery from every kind of live-but-hung task. No implementation changes were made during this investigation. - -## Reservation lifecycle - -### 1. Reserve before forwarding - -`pay_for_request()` in `routstr/auth.py` reserves funds before dispatching the billed request upstream. - -It creates a durable `ReservationRelease` identity containing: - -- `id`: the individual reservation identity; -- `key_hash`: the request's key; -- `billing_key_hash`: the key whose balance backs the request; -- `reserved_msats`: the amount owned by this reservation; -- `status`: initially `active`; -- `created_at`: initially the current timestamp. - -The aggregate reserved balance and durable reservation row commit together. The request's reservation identity matters: releasing one request must not erase funds reserved by another concurrent request. - -`ApiKey.reserved_at` is also stamped when funds are reserved. It is an aggregate timestamp, not an independent timestamp for each request. - -### 2. Renew while the owner task remains alive - -`_start_reservation_heartbeat()` in `routstr/auth.py` starts a task for each reservation. Its interval is: - -```python -max(1, settings.stale_reservation_timeout_seconds // 3) -``` - -With the default timeout of 300 seconds, renewal occurs approximately every 100 seconds. - -The heartbeat captures `asyncio.current_task()` as the owner. At each iteration it checks: - -```python -if owner is None or owner.done(): - return -``` - -If the owner is still alive, it calls `renew_reservation()` using a separate database session. Renewal updates the active durable row's `created_at` to the current time. - -Important consequences: - -- Renewal depends on task lifetime, not demonstrated request progress. -- There is no original-age limit in this heartbeat. -- `created_at` is overwritten, so it actually serves as a renewable lease timestamp. -- An owner that has finished cannot keep renewing indefinitely through this heartbeat. -- An owner that is blocked indefinitely may keep renewing indefinitely. - -### 3. Settle or release - -Normal completion settles the charge and releases the reservation. Handled upstream failures revert the reservation. Terminal reservation transitions stop the heartbeat. - -The proxy includes cancellation cleanup. Streaming paths use finalizers and ownership wrappers to improve cleanup across cancellation and downstream-send failures. Relevant code includes: - -- `routstr/auth.py`; -- `routstr/proxy.py`; -- `routstr/upstream/base.py`; -- `routstr/upstream/stream_ownership.py`. - -If a request dies without completing cleanup, its heartbeat is intended to stop once the owning task is done. The reservation can then age out and be released by the sweeper. - -## Existing cleanup mechanisms - -### Background sweep - -`periodic_stale_reservation_sweep()` in `routstr/auth.py` is started by the application lifespan in `routstr/core/main.py`. - -Defaults: - -| Setting/mechanism | Default | Meaning | -| --- | --- | --- | -| `STALE_RESERVATION_TIMEOUT_SECONDS` | 300 seconds | Maximum age of an unrenewed reservation lease before it is stale | -| `STALE_RESERVATION_SWEEP_INTERVAL_SECONDS` | 60 seconds | Interval between background cleanup passes | -| Heartbeat interval | 100 seconds | Approximately one third of the stale timeout | -| `UPSTREAM_READ_TIMEOUT` | 900 seconds | Upstream HTTP read inactivity timeout, not a total request deadline | -| `RESET_RESERVED_BALANCE_ON_STARTUP` | `True` | Explicit startup reset of active reservations and aggregate reserved balances | - -The sweeper calls `release_stale_reservations()` in `routstr/core/db.py`. - -For durable reservations, it selects `active` rows whose `created_at` is older than the cutoff. Its terminal update also checks the timestamp, protecting against a heartbeat that renews between selection and release. - -Each successful release subtracts that reservation's own amount from the relevant aggregates. Healthy releases commit individually so that certain later corruption repairs cannot roll them back. - -Under healthy execution, recovery occurs after the last lease renewal has aged beyond the configured timeout, plus sweep scheduling and database-operation time. This is **not** a guarantee of release 300 seconds after the request originally began. - -### Refund-time cleanup - -`refund_wallet_endpoint()` in `routstr/balance.py` checks for reserved funds before opening the refund claim. - -If `key.reserved_balance > 0`, it: - -1. Calls `release_stale_reservations()` scoped to that key. -2. Refreshes the key from the database. -3. Returns HTTP 400 with the reported message if reserved balance remains. - -Thus, the current refund path does not rely exclusively on the background task having run. A stale durable reservation should also be releasable during refund itself. - -If cleanup raises an unexpected exception instead, that is a separate failure from this specific HTTP 400 branch. - -### Legacy aggregate cleanup - -Older deployments may have aggregate reserved balances without matching durable rows. - -The cleanup function also looks for these legacy aggregates, but only clears them when there is no active durable owner. It uses a compare-and-swap guard on the observed balance and timestamp to avoid erasing a newly created reservation. - -The behavior differs between background and targeted cleanup: - -| Legacy aggregate state, with no active durable owner | Background sweep | Refund-time targeted cleanup | -| --- | --- | --- | -| Old `reserved_at` | Eligible for release | Eligible for release | -| Recent `reserved_at` | Preserved | Preserved | -| `reserved_at = NULL` | Deliberately skipped | Eligible for repair | - -The NULL-timestamp behavior is explicitly covered by existing tests. It is a background-recovery limitation, but **alone it does not explain the reported refund rejection on the current checkout**, because targeted refund cleanup heals it. - -### Startup reset - -When enabled, startup calls `reset_all_reserved_balances()`. It marks active durable reservations released and clears aggregate reserved balances and timestamps. - -This is not a safe universal operational fix. In a shared-database, multi-instance setup, another instance may still own a legitimate in-flight request. Resetting its reservation can break billing. The setting's source comment recommends disabling it for horizontal scaling. - -## Why the 900-second HTTP timeout does not guarantee eventual completion - -The user correctly asks: if the last request was days ago, shouldn't a 900-second upstream timeout have completed or failed the request long before now? - -**For an ordinary request actively waiting for upstream bytes, with no bytes arriving, yes.** It should hit the read timeout and reach failure cleanup. A days-long refund blockage is abnormal, not expected behavior for a silent upstream. - -However, the HTTP read timeout is not an absolute deadline spanning the complete request lifecycle. - -### Upstream continues sending bytes - -A stream can avoid a read inactivity timeout by delivering bytes periodically. Those bytes might be content or keepalive traffic. A stream with no total-duration limit could therefore remain open longer than 900 seconds. - -This is a technical possibility, **not evidence that the affected upstream streamed for days**. It must not be assumed as the production explanation. - -### Router is blocked writing to the downstream client - -If the router has received a chunk and is blocked delivering it to the client, it may not currently be waiting on an upstream HTTP read. The upstream read timeout is not a general bound on downstream ASGI sends. - -Whether a particular blocked send keeps the captured owner task alive depends on the execution path. That behavior needs a runtime trace or regression test, rather than an assumption about all stream paths. - -### Router is blocked after upstream completion - -Database settlement, finalization, or resource cleanup happens outside the upstream read operation. The upstream read timeout does not bound these waits. - -If the heartbeat's owning task remains alive while waiting, renewal may continue. If that owner finishes and only detached cleanup remains, the heartbeat should stop and the sweeper should eventually recover the reservation. - -### Conclusion - -The current code has no common hard lifetime limit found in this investigation that covers reservation creation, upstream dispatch, streaming delivery, and finalization together. - -The missing guarantee is: - -> A live-but-stuck request cannot renew its reservation forever. - -This gap is confirmed by the heartbeat's renewal condition. The specific blocked operation, if any, on the affected node is not known. - -## Findings and hypotheses - -### Confirmed: renewal does not require progress - -An owner task being alive is sufficient to renew the lease. Neither original request age nor meaningful progress is checked. - -This permits indefinite reservation retention in principle, even without new requests using the key. - -### Confirmed: immutable request age is not stored in the reservation row - -`ReservationRelease.created_at` doubles as the last-renewal timestamp. Once renewed, it cannot tell us when the request originally started. - -This impairs diagnostics and prevents enforcing an original-age limit from this field alone. - -### Confirmed: NULL legacy timestamps are not background-cleaned - -Such keys may remain reserved indefinitely in the background. The current refund endpoint has targeted recovery for this state, subject to the absence of an active durable owner. - -### Confirmed: unexpected failures can interrupt a sweep pass - -The background loop catches unexpected exceptions, logs `Error in periodic_stale_reservation_sweep`, and retries after the sweep interval. - -Some aggregate-corruption cases are handled per reservation, but not every database exception is isolated per record. A persistently failing operation could repeatedly interrupt a pass. Whether this prevents a particular key's cleanup depends on the failure and processing order. - -There is no evidence yet that this caused the reported error. - -### Possible: affected deployment differs from this checkout - -The current code includes heartbeat-owner binding, targeted legacy recovery, and corruption handling. The affected node may run older or different code. - -The deployed commit must be established before treating local behavior as proof of production behavior. - -### Possible: future timestamps or unusual effective configuration - -A future-dated lease can remain non-stale unexpectedly. An unusually large configured timeout can also preserve old reservations. - -Clock skew between instances sharing a database can affect lease timestamps and age calculations. These are diagnostic checks, not confirmed causes. - -## Existing verified recovery coverage - -The suites run during this investigation cover, among other cases: - -- Stamping aggregate reservation timestamps on payment. -- Reverting individual reservations without erasing siblings. -- Releasing old reservations and preserving fresh ones. -- Resetting reserved balances during explicit startup reset. -- Refund-time recovery of stale and legacy NULL-timestamp aggregates. -- Refusing refunds while a recent reservation remains. -- Streaming finalization and client-disconnect cleanup. -- Owner task termination allowing recovery of an abandoned reservation. -- Lease renewal across an in-flight request. -- Renewal racing with stale release. -- Legacy aggregate release racing with a new reservation. -- Several accounting-corruption cases and safe terminal repair. -- Preventing late charges after a reservation has reached a released terminal state. - -These tests do not substitute for explicit tests of endless keepalive streams, blocked downstream sends, or finalization that never completes. - -## Production diagnosis: distinguish a renewing lease from failed cleanup - -The most useful initial question is: - -> Is the reservation still being renewed, or is it stale and not being released? - -Do not share the raw API-key secret. Use its stored hash and reservation identifiers in restricted operational diagnostics. - -### 1. Establish deployment and configuration - -Record: - -- Deployed commit/version. -- Effective `STALE_RESERVATION_TIMEOUT_SECONDS`. -- Effective `UPSTREAM_READ_TIMEOUT`. -- Startup-reset setting. -- Number of instances sharing the database. -- Current time on each relevant instance. -- Whether the lifespan/background tasks completed startup. - -Use effective settings, not only environment variables; settings initialization includes persisted configuration. - -### 2. Inspect the key and all related reservations - -Read-only queries: - -```sql -SELECT hashed_key, balance, reserved_balance, reserved_at -FROM api_keys -WHERE hashed_key = :key_hash; - -SELECT id, key_hash, billing_key_hash, - reserved_msats, status, created_at -FROM reservation_releases -WHERE key_hash = :key_hash - OR billing_key_hash = :key_hash; -``` - -Inspect both key relationships, since a reservation may reference the key as request owner or billing owner. - -Take two snapshots approximately 110 seconds apart with default settings, or use an interval longer than the effective heartbeat interval. A pair of snapshots is a useful signal; it is not a substitute for longer observation when renewal is delayed or intermittent. - -### 3. Interpret the results - -| Observation | Investigation direction | -| --- | --- | -| Active reservation timestamp advances | Identify the instance and owning task renewing it; inspect its stack and actual progress | -| Active reservation timestamp is older than the stale cutoff and does not advance | Check sweep execution/errors, refund cleanup, deployed code, and accounting state | -| Reserved balance remains with no active durable rows | Inspect legacy timestamp and aggregate recovery; current targeted refund cleanup should repair stale/NULL state | -| Lease timestamp is in the future | Check clocks and timestamp integrity | -| Some rows are stale and others fresh | Release only stale owners; do not clear the whole key | -| Aggregate amount disagrees with active durable ownership | Investigate accounting drift and safe reconciliation | - -If the lease is genuinely days old and unrenewed, the indefinite-heartbeat explanation does **not** explain that row. Cleanup failure or incompatible deployment becomes the relevant direction. - -### 4. Inspect logs and task state - -Relevant existing log messages include: - -- `Error in periodic_stale_reservation_sweep`. -- `Failed to renew billing reservation lease`. -- `Released stale reservations`. -- `Released corrupt stale reservation without aggregate subtraction`. -- `Released corrupt reservation without aggregate subtraction`. -- `Client disconnected mid-request, reverting reservation`. -- `refund_wallet_endpoint: released stale reservation before refund`. - -For a renewing lease, locate the process with that reservation's heartbeat and inspect the owner's stack. Determine whether it is waiting on upstream input, downstream delivery, database work, finalization, or another operation. - -Also correlate the original request with upstream outcome and billing logs. A heartbeat alone does not demonstrate that inference is still running. - -## Proposed hardening - -These are proposed changes, not completed fixes. - -### 1. Separate original age from renewable lease age - -Keep distinct durable fields for: - -- Immutable reservation/request start time. -- Last lease renewal time. - -Consider additional progress and ownership metadata where justified. Define migration behavior explicitly: existing renewed `created_at` values cannot reconstruct true original start times. - -### 2. Bound the actual request, not just the accounting lease - -Introduce a configurable total billed-request lifetime covering all relevant routes and phases, including streaming delivery. Add appropriate inactivity bounds for upstream waits and downstream delivery, and bounded finalization/cleanup behavior. - -Timeout handling should: - -1. Stop or cancel the owning request and close owned resources. -2. Settle known or estimated delivered usage according to existing billing policy. -3. Release only that request's remaining reservation. -4. Stop heartbeat renewal. -5. Reach a durable terminal state that prevents later charging. - -Do **not** merely stop renewal or zero the key while a request continues running. Releasing funds while upstream work can still finish creates refund/late-charge and provider-cost risks. - -Care is also needed not to cancel legitimate long-running inference accidentally. Request lifetime, inactivity, and lease expiry are different concepts and should have distinct documented policies. - -### 3. Improve stalled-owner detection and observability - -Expose actionable, non-secret diagnostics: - -- Reservation identity and owning instance. -- Immutable age and current lease age. -- Last meaningful progress and current phase, if tracked. -- Reason for terminal transition or refused refund. -- Age and count of active reservations. -- Sweep failures and cleanup duration. - -Do not treat upstream keepalive bytes as necessarily meaningful model progress. Decide deliberately which signals should extend which deadlines. - -### 4. Reconcile legacy and inconsistent aggregates safely - -Define a migration/recovery policy for NULL legacy timestamps, rather than leaving them background-ineligible indefinitely. - -Mixed-version deployments require caution: an aggregate without a durable row might still belong to an older live worker. Any reconciliation must preserve valid durable owners and avoid unsafe whole-key resets. - -Investigate positive residual aggregates even after durable rows become terminal, with concurrency guards and accounting invariants preserved. - -### 5. Make cleanup failures diagnosable and resilient - -Consider bounded database operations, per-record failure isolation where safe, and alerts for repeated sweep failures or reservations exceeding expected age. - -Failure isolation must not weaken atomicity between durable transitions and aggregate updates. A failed release must not partially debit unrelated reservations. - -## Regression tests needed to close the gaps - -Add tests that reproduce and verify recovery for: - -1. A live owner waiting indefinitely without progress. -2. An endless upstream stream sending keepalive bytes below the read-timeout interval. -3. A downstream send blocked indefinitely after receiving an upstream chunk. -4. Finalization or database settlement that stalls. -5. Cancellation before streaming begins, during streaming, and during finalization. -6. Renewing lease older than the new maximum original-age limit. -7. Background legacy NULL-timestamp recovery under the chosen migration policy. -8. Corrupt residual aggregates alongside a healthy active sibling reservation. -9. A failing cleanup operation followed by other recoverable reservations. -10. Multiple workers concurrently renewing, sweeping, timing out, and refunding. -11. Late completion attempting to charge after timeout/release. -12. Future lease timestamps and the chosen clock-skew policy. - -For each timeout/recovery test, assert: - -- The underlying request/resource is stopped or closed as intended. -- No heartbeat can renew indefinitely afterward. -- Only the affected reservation is released. -- Sibling reservations remain intact. -- Balance/reserved accounting remains valid. -- Terminal transitions are idempotent. -- A later completion cannot charge released/refunded funds. -- The key becomes refundable when no legitimate reservations remain. - -## Operational caution - -Do not solve the symptom by manually setting `reserved_balance = 0` while active requests or heartbeat tasks may exist. Durable reservation state and aggregate balances must agree, and late completion must not be allowed to spend refunded funds. - -Any production repair should begin with a read-only snapshot and identification of live ownership, then use an accounting-safe terminal transition or controlled maintenance procedure. - -## Bottom line - -The expected stale cleanup exists. A genuinely dead, unrenewed reservation should recover on the current version with healthy database access, including during a refund attempt. - -The confirmed design gap is that **a task remaining alive is sufficient to renew its reservation indefinitely**, and the upstream 900-second read timeout does not bound every phase of that task's lifetime. - -A days-old refund blockage therefore warrants investigation, not an assumption that normal request processing is still underway. The first decisive evidence is whether the affected reservation's lease timestamp continues advancing. The production root cause and implementation fixes remain open. - -## Release-specific reproduction: v0.4.7 (confirmed) - -The user subsequently confirmed that the affected node runs the released **v0.4.7** tag. Testing that tag revealed an important correction to the initial analysis above: - -**The 900-second upstream read timeout exists in the newer checkout, not in v0.4.7.** The release's forwarding paths construct `httpx.AsyncClient(..., timeout=None)`. It has no `upstream_read_timeout` settings field. Setting `UPSTREAM_READ_TIMEOUT=3` in the reproduction did nothing; importing the release settings confirmed the field is absent. - -Therefore, on this release an upstream can send one chunk and then remain completely silent without triggering an HTTP read timeout. Periodic bytes are not needed to explain indefinite waiting. - -### Environment and isolation - -- Podman: 5.8.4, netavark network backend. -- Release commit: `f32565e2547abbbffd77a01198ef683ecb8e3d4f`. -- Detached worktree: `.worktrees/reserved-balance-v047`. -- Built the release's own Dockerfile (Python 3.11 base), without source patches. -- Image: `localhost/routstr-reserved-repro:v0.4.7`. -- Image ID: `8c3340a33040df37439f7085e369050d3acc2adf0324bc024e1b4288a3601f76`. -- Separate containers, loopback ports 18080/18081, container-local SQLite database. -- No original node database, wallet, secrets, volumes, or image tag were changed. -- Host networking avoided the reported aardvark DNS issue for this experiment; containerized DNS was not tested or repaired. -- Accelerated stale timeout: 6 seconds, heartbeat every 2 seconds. Background sweep retained its actual 60-second interval. -- Synthetic database-funded keys avoided introducing Cashu mint behavior into the reservation test. Actual refund payout success was not tested. - -### Dummy upstream scenarios - -A small local OpenAI-compatible server exposed `/v1/models` and `/v1/chat/completions` using `gpt-4o-mini`: - -1. **Finite:** three chunks, a usage event, and `[DONE]`. -2. **Silent:** one chunk, then sleep for 3600 seconds. -3. **Endless:** a content chunk every 0.5 seconds with no terminal event. - -The test client consumed streams, queried reservation state, attempted refunds, and disconnected. Evidence and reusable scripts are in `reservation-repro-v047/`. - -### Observed results - -| Scenario | Outcome | -| --- | --- | -| Finite stream | Settled normally; reserved balance became zero | -| Silent stream | Did not time out; durable lease kept renewing | -| Endless stream | Lease kept renewing; refund returned the exact reported HTTP 400 | -| Both clients disconnected | Both upstream connections remained established; both reservations remained active and kept renewing | -| After more than a background-sweep interval | The abandoned reservations were still active; their fresh leases prevented stale cleanup | -| Dummy upstream forcibly stopped | Both requests finally reached error/finalization; both reserved balances became zero and rows became `charged` | - -Both streams reserved 11 msats. Their lease timestamps initially advanced from `1790765677` through `1790765685` and `1790765695`. After client termination, a later snapshot at `1790765791` still showed both rows `active` with leases at `1790765789`. This is approximately 116 seconds after their creation and well beyond the accelerated stale timeout and a background-sweep interval. - -At that later point, `ss` showed two established router-to-upstream connections and no test-client connection on port 18080. A refund for the silent key still returned: - -```json -{"detail":"Cannot refund key. There are ongoing requests for this api key."} -``` - -Stopping the dummy upstream broke those connections. Finalization then charged estimated usage and cleared the reservations. The finite and silent keys ended with a 3-msat charge; the endless stream accumulated a 25-msat charge. This also demonstrates that abandoned upstream work can continue affecting billing after the downstream client is gone. - -The first probe run ended with a client-side `TimeoutError` because it expected the silent stream to complete. That timeout was imposed by the probe's `asyncio.wait_for`, not by the router. The saved probe was subsequently adjusted to report this expected observation rather than crash. - -### What this establishes - -We have reproduced a plausible mechanism for a key remaining blocked long after the client last used it on **the exact release tag**: - -1. The upstream stream remains open, even silently. -2. Downstream disconnection does not terminate the upstream-owning request in the tested runtime/path. -3. The owner remains alive, so its heartbeat keeps renewing. -4. Background and refund-time stale cleanup preserve the fresh lease. -5. Refund remains blocked indefinitely unless the upstream closes or another intervention stops the owning work. - -The reproduction lasted minutes, not days. The absence of a read timeout and continuing renewal explain how the state can persist longer; no days-long run was performed. - -This is concrete release-specific evidence, but not proof that the affected production key has this exact state. Production confirmation still requires reservation snapshots and logs. - -### Shutdown symptoms - -The dummy upstream also needed SIGKILL after a short SIGTERM grace period while its streams were open. The router stopped normally after the upstream was stopped and its streams finalized. - -This supports the possibility that outstanding streaming work can delay graceful shutdown. It does not establish that the user's earlier router/UI shutdown warnings share the same cause. The aardvark DNS removal failure is a separate Podman networking symptom; the reproduction does not require it. - -### Next implementation work - -Prioritize fixes/backports appropriate to v0.4.7: - -- Finite upstream transport timeouts, including reads and header waits. -- Reliable downstream-disconnect propagation and deterministic closure/finalization of owned streaming resources in the deployed FastAPI/Starlette/Uvicorn combination. -- A maximum request lifetime independent of renewable leases and keepalive bytes. -- Real-network regression tests that disconnect a client from a silent upstream stream and assert upstream closure, terminal billing state, stopped renewal, and zero residual reservation. - -The newer checkout has transport timeout and stream-ownership changes, but this experiment did not validate the same scenario against that newer checkout. Do not assume an upgrade fully fixes every gap without rerunning the reproduction. - -Both reproduction containers were stopped at the end. Their container-local database and logs were retained for inspection; no original services were restarted. - -## Current main reproduction: timeout does not close every gap - -The same investigation was repeated against unpatched local main commit `96c8e2f77de8e9f8a0979d17dba0a6d20c78fe89` using its own Dockerfile and frozen dependencies. The main image ran Python 3.14, Starlette 1.6.0, and Uvicorn 0.31.1. Detailed commands and evidence are in `reservation-repro-main/README.md`. - -With an effective upstream read timeout of 3 seconds and stale timeout of 6 seconds: - -- Finite completion settled correctly. -- Silent upstream streams reached the read timeout and cleared reservations. -- A header wait timed out with HTTP 424 and released its reservation. -- An endless content stream **after client disconnect** kept renewing and returning the reported refund HTTP 400. -- SSE comment-only keepalives evaded read timeout; renewal persisted even after disconnect. -- A flood stream to a downstream client that never read remained reserved, including after its socket closed. - -The three problematic keys remained active approximately 269 seconds after request start, across multiple background sweeps, with fresh lease timestamps. This is not merely an active client asking for a refund: all downstream test clients were gone well before the final observation. - -### Framework compatibility concern - -Installed framework source provides a specific lead: - -- Uvicorn's httptools protocol advertises ASGI HTTP 2.4. -- Its send function silently returns after downstream disconnection. -- Starlette's ASGI >=2.4 StreamingResponse path expects send to raise OSError for disconnect detection and does not run the older disconnect listener. - -This mismatch is consistent with streams continuing to consume upstream bytes while downstream sends become no-ops. Captured code is in the evidence directory. A runtime task-stack or controlled framework-version comparison is still needed for complete causal validation. - -Removing only LoggingMiddleware in a diagnostic router did not resolve disconnect renewal. Therefore, do not attribute the disconnect problem solely to that middleware. - -### Additional finalization/shutdown observation - -Forcibly stopping the dummy upstream finalized the diagnostic router's streams, but the unmodified router still had active reservations five seconds after upstream termination and required SIGKILL after a ten-second SIGTERM grace period. Its logs showed upstream termination warnings without completed settlement for those three requests in the captured window. The precise blocked operation was not traced. - -This adds a finalization/delivery investigation beyond transport inactivity. In this main reproduction, unlike the release reproduction, upstream termination did not promptly clear every reservation. - -### Updated conclusion - -The newer read timeout fixes silent upstream waits, but **does not eliminate reservation leaks for disconnected clients whose upstream streams keep producing bytes, or stalled downstream delivery**. The stream ownership/finalizer unit tests previously run do not exercise the complete real server/framework/middleware network path that exposed these cases. - -Prioritize real-network regression coverage and disconnect propagation, the installed server/framework compatibility, bounded downstream delivery and finalization, and an absolute request lifetime independent of keepalive traffic. No implementation fix has been made; alternate routes, multi-worker behavior, and database fault injection remain untested. diff --git a/pyproject.toml b/pyproject.toml index 1faf6828..7daf7489 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -73,7 +73,7 @@ build-backend = "setuptools.build_meta" packages = ["routstr"] [tool.ruff] -extend-exclude = ["examples", "repro", "reservation-repro-main"] +extend-exclude = ["examples"] [tool.ruff.lint] select = ["E", "F", "I"] @@ -87,7 +87,6 @@ check_untyped_defs = true disallow_untyped_calls = true disallow_incomplete_defs = true disallow_untyped_decorators = true -exclude = ["^repro/", "^reservation-repro-main/"] [tool.uv.sources] routstr = { workspace = true } diff --git a/repro/IMPLEMENTATION.md b/repro/IMPLEMENTATION.md deleted file mode 100644 index 1c231f1f..00000000 --- a/repro/IMPLEMENTATION.md +++ /dev/null @@ -1,46 +0,0 @@ -# Reservation lifecycle implementation and validation - -Branch: fix/reservation-lifecycle. Baseline: 96c8e2f7. - -## Implemented - -- Outermost pure-ASGI lifecycle supervision with one coordinated receive consumer, explicit disconnect monitoring, cancellation, and exact reservation fallback cleanup. -- Finite overall request lifetime (MAX_REQUEST_LIFETIME_SECONDS, default 1800), downstream send timeout (DOWNSTREAM_SEND_TIMEOUT_SECONDS, default 60), and cleanup timeout (REQUEST_CLEANUP_TIMEOUT_SECONDS, default 30). -- Lifecycle identity shared through context across middleware tasks; reservation replacements are registered for exact cleanup. -- Heartbeats stop on lifecycle termination or local maximum age. -- Persistent stream finalization has a finite cleanup budget. -- Durable immutable started_at and expires_at columns; expiry covers remaining request lifetime plus settlement grace, including provider fallback without restarting the original deadline. -- Renewal and charge claims refuse expired reservations. Sweeping can release absolute-expired reservations even when their renewable timestamp is fresh. -- Migration grants legacy active rows 1830 seconds of grace; original ages are not fabricated. Drain old workers before deployment. - -## Verification - -Run from worktree with PYTHONPATH=$PWD because the shared root virtual environment's editable install points at the original checkout: - -PYTHONPATH=$PWD ../../.venv/bin/pytest tests/unit/test_request_lifecycle.py tests/unit/test_stale_reservations.py tests/unit/test_streaming_billing_finalization.py tests/integration/test_negative_available_balance_repro.py -q - -64 tests passed. Ruff checks passed on changed files. Full-project mypy was attempted but did not finish within the tool timeout; no successful typecheck is claimed. - -Final built image: localhost/routstr-reserved-repro:fix, ef81426ad79e3d14ec462a39ab1f7481fd0cb410a9cb93cb42de41e8b3523869. - -Container tests used real TCP, full middleware stack, frozen image dependencies, isolated SQLite and synthetic balances. Read timeout 3s, lifetime 15s, delivery timeout 2s, cleanup timeout 3s, stale timeout 6s. - -Reused the main probe on ports 18100/18101. Results in results-final.txt and router-final.log: - -- Finite and silent streams settled. -- Header wait released its reservation. -- Disconnected endless stream no longer retained its reservation. -- Non-reading flood client hit bounded delivery/cleanup. -- Connected keepalive-only stream terminated at maximum lifetime. -- After the background-sweep interval and all client closures: every key reserved_balance=0, no active durable reservations. Explicit database assertions passed. -- Router shut down within the 10-second grace without SIGKILL. Dummy upstream still required SIGKILL: its fixture deliberately sleeps/open-streams and is not patched router code. - -Actual mint payout was not tested. Protocol errors on already-started streams when deadlines interrupt them are expected; an HTTP status cannot be replaced after headers are sent. - -## Financial policy / limitations - -The lifecycle first lets existing finalization run within a bounded budget. If still active, fallback releases only that reservation; late charge is fenced by terminal state. This can forgo charging observed output on failed settlement. It prioritizes freeing customer funds over leaving them locked; review this policy before deployment. Upstream compute may continue remotely even after local connection closure. - -This implementation does not complete every proposed hardening idea: provider cancellation APIs, full observability, per-record unexpected DB-failure isolation, legacy NULL aggregate background reconciliation, multi-worker/alternate-route network matrix and DB-outage injection remain follow-up work. No dependency upgrade was needed for the tested cases because explicit disconnect supervision avoids relying solely on send errors. - -All reproduction containers are stopped. Original node data/configuration is untouched. Source changes are uncommitted in the worktree for review. diff --git a/repro/dummy_upstream.py b/repro/dummy_upstream.py deleted file mode 100644 index 1be75ecc..00000000 --- a/repro/dummy_upstream.py +++ /dev/null @@ -1,45 +0,0 @@ -"""Loopback-only streaming fixture; no router monkeypatches.""" -import asyncio -import json -import time -from fastapi import FastAPI, Request -from fastapi.responses import StreamingResponse - -app = FastAPI() -events = [] - -@app.get('/events') -async def history(): - return events - -@app.get('/v1/models') -async def models(): - return {'object': 'list', 'data': [{'id': 'gpt-4o-mini', 'object': 'model', 'created': 1, 'owned_by': 'repro'}]} - -@app.post('/v1/chat/completions') -async def completions(request: Request): - body = await request.json() - mode = body.get('messages', [{}])[0].get('content', 'finite') - events.append({'event': 'start', 'mode': mode, 'time': time.time()}) - if mode.startswith('header'): - await asyncio.sleep(3600) - async def stream(): - count = 0 - try: - while True: - if mode.startswith('keepalive'): - yield ': ping\n\n' - else: - chunk = {'id': 'repro', 'object': 'chat.completion.chunk', 'created': int(time.time()), 'model': 'gpt-4o-mini', 'choices': [{'index': 0, 'delta': {'content': 'x' * (65536 if mode.startswith('flood') else 1)}, 'finish_reason': None}]} - yield 'data: ' + json.dumps(chunk) + '\n\n' - count += 1 - if mode == 'finite' and count >= 3: - yield 'data: ' + json.dumps({'id': 'repro', 'object': 'chat.completion.chunk', 'model': 'gpt-4o-mini', 'choices': [], 'usage': {'prompt_tokens': 1, 'completion_tokens': count, 'total_tokens': count + 1}}) + '\n\n' - yield 'data: [DONE]\n\n' - return - await asyncio.sleep(3600 if mode.startswith('silent') else (0.001 if mode.startswith('flood') else 0.5)) - finally: - event = {'event': 'close', 'mode': mode, 'chunks': count, 'time': time.time()} - events.append(event) - print(json.dumps(event), flush=True) - return StreamingResponse(stream(), media_type='text/event-stream') diff --git a/repro/probe.py b/repro/probe.py deleted file mode 100644 index 93d71a75..00000000 --- a/repro/probe.py +++ /dev/null @@ -1,60 +0,0 @@ -import asyncio -import json -import socket -import subprocess -import time -import httpx - -BASE='http://127.0.0.1:18100' - -def snapshot(): - code="import sqlite3,json,time; c=sqlite3.connect('/tmp/reserved-fix.db'); c.row_factory=sqlite3.Row; print(json.dumps({'time':time.time(),'keys':[dict(r) for r in c.execute(\"select hashed_key,balance,reserved_balance,reserved_at from api_keys where hashed_key like 'main-%'\")],'rows':[dict(r) for r in c.execute(\"select * from reservation_releases where key_hash like 'main-%'\")]}))" - return json.loads(subprocess.check_output(['podman','exec','reserved-router-fix','/.venv/bin/python','-c',code],text=True)) - -async def consume(mode): - try: - async with httpx.AsyncClient(timeout=None) as c: - async with c.stream('POST',BASE+'/v1/chat/completions',headers={'Authorization':'Bearer sk-main-'+mode},json={'model':'gpt-4o-mini','messages':[{'role':'user','content':mode}],'stream':True,'max_tokens':10}) as r: - print('STREAM',mode,r.status_code,flush=True) - async for _ in r.aiter_bytes(): pass - print('ENDED',mode,flush=True) - except asyncio.CancelledError: - print('CLIENT_DISCONNECTED',mode,flush=True) - raise - except Exception as e: - print('CLIENT_ERROR',mode,type(e).__name__,str(e),flush=True) - -async def report(label): - print(label,json.dumps(snapshot()),flush=True) - async with httpx.AsyncClient(timeout=5) as c: - for mode in ['silent-disconnect','endless-disconnect','keepalive','flood','header']: - # Only attempt payout while reserved: avoid requiring a real mint. - if next(k for k in snapshot()['keys'] if k['hashed_key']=='main-'+mode)['reserved_balance']: - r=await c.post(BASE+'/v1/wallet/refund',headers={'Authorization':'Bearer sk-main-'+mode}) - print('REFUND',mode,r.status_code,r.text,flush=True) - print('UPSTREAM_EVENTS',json.dumps((await c.get('http://127.0.0.1:18101/events')).json()),flush=True) - -async def main(): - modes=['finite','silent','silent-disconnect','endless-disconnect','keepalive','header'] - tasks={m:asyncio.create_task(consume(m)) for m in modes} - # Real client with a small receive buffer, never draining the HTTP response. - sock=socket.socket(); sock.setsockopt(socket.SOL_SOCKET,socket.SO_RCVBUF,1024); sock.connect(('127.0.0.1',18100)) - body=json.dumps({'model':'gpt-4o-mini','messages':[{'role':'user','content':'flood'}],'stream':True,'max_tokens':10}).encode() - sock.sendall(b'POST /v1/chat/completions HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer sk-main-flood\r\nContent-Type: application/json\r\nContent-Length: '+str(len(body)).encode()+b'\r\n\r\n'+body) - await asyncio.sleep(1) - for m in ['silent-disconnect','endless-disconnect']: - tasks[m].cancel() - await asyncio.gather(tasks['silent-disconnect'],tasks['endless-disconnect'],return_exceptions=True) - await asyncio.sleep(9) - await report('AT_10_SECONDS') - await asyncio.sleep(60) - await report('AFTER_SWEEP') - sock.close() - tasks['keepalive'].cancel() - await asyncio.gather(tasks['keepalive'],return_exceptions=True) - await asyncio.sleep(8) - await report('AFTER_ALL_CLIENTS_CLOSED') - for task in tasks.values(): task.cancel() - await asyncio.gather(*tasks.values(),return_exceptions=True) - -asyncio.run(main()) diff --git a/repro/results-final.txt b/repro/results-final.txt deleted file mode 100644 index 6545bf6d..00000000 --- a/repro/results-final.txt +++ /dev/null @@ -1,19 +0,0 @@ -STREAM endless-disconnect 200 -STREAM keepalive 200 -STREAM finite 200 -STREAM silent 200 -STREAM silent-disconnect 200 -CLIENT_DISCONNECTED silent-disconnect -CLIENT_DISCONNECTED endless-disconnect -ENDED finite -ENDED silent -STREAM header 424 -ENDED header -AT_10_SECONDS {"time": 1790767971.1792026, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 12, "reserved_at": 1790767960}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "9b5b9063f6f045a290611e4144c897de", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "released", "created_at": 1790767962, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "efc8563301ba458890ee3ab2335e4e49", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cbed89b7ec444fea9789cf53b3e0f476", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "active", "created_at": 1790767971, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1c56b25265df4743b4cff70dc57544c6", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cc7f113eef004b7ba27bc761c5d9b9a1", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "6a1b5506b8ab4747884475075ebc38da", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767961, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1dfedc1c2cee407fb2cd1a28b1253c6d", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}]} -REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"45f783cc-4c0b-4222-be13-8e726fc7cebc"} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}, {"event": "start", "mode": "flood", "time": 1790767960.9594278}, {"event": "start", "mode": "endless-disconnect", "time": 1790767961.008998}, {"event": "start", "mode": "keepalive", "time": 1790767961.036595}, {"event": "start", "mode": "finite", "time": 1790767961.060875}, {"event": "start", "mode": "silent", "time": 1790767961.0869172}, {"event": "start", "mode": "silent-disconnect", "time": 1790767961.1580715}, {"event": "start", "mode": "header", "time": 1790767961.1944675}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767962.064207}] -CLIENT_ERROR keepalive RemoteProtocolError peer closed connection without sending complete message body (incomplete chunked read) -AFTER_SWEEP {"time": 1790768033.9280283, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "9b5b9063f6f045a290611e4144c897de", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "released", "created_at": 1790767962, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "efc8563301ba458890ee3ab2335e4e49", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cbed89b7ec444fea9789cf53b3e0f476", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767975, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1c56b25265df4743b4cff70dc57544c6", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cc7f113eef004b7ba27bc761c5d9b9a1", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "6a1b5506b8ab4747884475075ebc38da", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767961, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1dfedc1c2cee407fb2cd1a28b1253c6d", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}]} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}, {"event": "start", "mode": "flood", "time": 1790767960.9594278}, {"event": "start", "mode": "endless-disconnect", "time": 1790767961.008998}, {"event": "start", "mode": "keepalive", "time": 1790767961.036595}, {"event": "start", "mode": "finite", "time": 1790767961.060875}, {"event": "start", "mode": "silent", "time": 1790767961.0869172}, {"event": "start", "mode": "silent-disconnect", "time": 1790767961.1580715}, {"event": "start", "mode": "header", "time": 1790767961.1944675}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767962.064207}] -AFTER_ALL_CLIENTS_CLOSED {"time": 1790768044.521246, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "9b5b9063f6f045a290611e4144c897de", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "released", "created_at": 1790767962, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "efc8563301ba458890ee3ab2335e4e49", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cbed89b7ec444fea9789cf53b3e0f476", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767975, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1c56b25265df4743b4cff70dc57544c6", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767960, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "cc7f113eef004b7ba27bc761c5d9b9a1", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "6a1b5506b8ab4747884475075ebc38da", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767961, "started_at": 1790767960, "expires_at": 1790767978}, {"id": "1dfedc1c2cee407fb2cd1a28b1253c6d", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767963, "started_at": 1790767960, "expires_at": 1790767978}]} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}, {"event": "start", "mode": "flood", "time": 1790767960.9594278}, {"event": "start", "mode": "endless-disconnect", "time": 1790767961.008998}, {"event": "start", "mode": "keepalive", "time": 1790767961.036595}, {"event": "start", "mode": "finite", "time": 1790767961.060875}, {"event": "start", "mode": "silent", "time": 1790767961.0869172}, {"event": "start", "mode": "silent-disconnect", "time": 1790767961.1580715}, {"event": "start", "mode": "header", "time": 1790767961.1944675}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767962.064207}] diff --git a/repro/results.txt b/repro/results.txt deleted file mode 100644 index 01417f00..00000000 --- a/repro/results.txt +++ /dev/null @@ -1,19 +0,0 @@ -STREAM silent-disconnect 200 -STREAM keepalive 200 -STREAM endless-disconnect 200 -STREAM silent 200 -STREAM finite 200 -CLIENT_DISCONNECTED silent-disconnect -CLIENT_DISCONNECTED endless-disconnect -ENDED finite -STREAM header 424 -ENDED header -ENDED silent -AT_10_SECONDS {"time": 1790767727.5333533, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 12, "reserved_at": 1790767717}, {"hashed_key": "main-flood", "balance": 999893370, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "79c7feb2274140748da2a97180f56d2c", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ecf20c01870b4ce49fc81bd300ee35df", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ec6f854d6a8647ac8b3bba50752bc647", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "active", "created_at": 1790767725, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "6da67d4c5cbd4ab19cc30f2f0fac6aaa", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "f4a35c5e89c343bfbe21415dce28a4d5", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "50a37cffba7745fa84d03b4070c86066", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "63ca23fa71024428baaf8baeb7ee3ede", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}]} -REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"f0bcc404-dffe-4860-8091-308f721ba053"} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}] -CLIENT_ERROR keepalive RemoteProtocolError peer closed connection without sending complete message body (incomplete chunked read) -AFTER_SWEEP {"time": 1790767790.5300848, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 999893370, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "79c7feb2274140748da2a97180f56d2c", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ecf20c01870b4ce49fc81bd300ee35df", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ec6f854d6a8647ac8b3bba50752bc647", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767731, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "6da67d4c5cbd4ab19cc30f2f0fac6aaa", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "f4a35c5e89c343bfbe21415dce28a4d5", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "50a37cffba7745fa84d03b4070c86066", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "63ca23fa71024428baaf8baeb7ee3ede", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}]} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}] -AFTER_ALL_CLIENTS_CLOSED {"time": 1790767800.920076, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-flood", "balance": 999893370, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "79c7feb2274140748da2a97180f56d2c", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ecf20c01870b4ce49fc81bd300ee35df", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 13, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "ec6f854d6a8647ac8b3bba50752bc647", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 12, "status": "released", "created_at": 1790767731, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "6da67d4c5cbd4ab19cc30f2f0fac6aaa", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 12, "status": "released", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "f4a35c5e89c343bfbe21415dce28a4d5", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 13, "status": "released", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "50a37cffba7745fa84d03b4070c86066", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 12, "status": "charged", "created_at": 1790767719, "started_at": 1790767717, "expires_at": 1790767735}, {"id": "63ca23fa71024428baaf8baeb7ee3ede", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 12, "status": "charged", "created_at": 1790767717, "started_at": 1790767717, "expires_at": 1790767735}]} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790767717.2871263}, {"event": "start", "mode": "silent-disconnect", "time": 1790767717.2960703}, {"event": "start", "mode": "keepalive", "time": 1790767717.3026786}, {"event": "start", "mode": "header", "time": 1790767717.3104746}, {"event": "start", "mode": "endless-disconnect", "time": 1790767717.318341}, {"event": "start", "mode": "silent", "time": 1790767717.3653235}, {"event": "start", "mode": "finite", "time": 1790767717.3924189}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790767718.397855}] diff --git a/repro/router-final.log b/repro/router-final.log deleted file mode 100644 index dee5fe67..00000000 --- a/repro/router-final.log +++ /dev/null @@ -1,108 +0,0 @@ -/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block - return result -2026-09-30 11:32:06 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). -2026-09-30 11:32:06 INFO uvicorn.error Started server process [1] -2026-09-30 11:32:06 INFO uvicorn.error Waiting for application startup. -2026-09-30 11:32:06 INFO routstr.core.main Application startup initiated -2026-09-30 11:32:10 INFO routstr.core.db Database migrations completed successfully -2026-09-30 11:32:10 INFO routstr.core.db Reset reserved balances on startup -2026-09-30 11:32:11 INFO routstr.upstream.helpers Seeding custom provider -2026-09-30 11:32:11 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings -2026-09-30 11:32:12 INFO routstr.proxy Initialized 1 upstream providers -2026-09-30 11:32:12 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider -2026-09-30 11:32:12 INFO routstr.nostr.analytics Usage analytics sharing task started -2026-09-30 11:32:12 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr -2026-09-30 11:32:12 INFO routstr.auth Dead-key pruning disabled (interval <= 0) -2026-09-30 11:32:12 INFO uvicorn.error Application startup complete. -2026-09-30 11:32:12 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18100 (Press CTRL+C to quit) -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Existing sk- API key found -2026-09-30 11:32:40 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:32:40 INFO routstr.auth Processing payment for request -2026-09-30 11:32:40 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:40 INFO routstr.payments RESERVE -2026-09-30 11:32:40 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:40 INFO routstr.payments RESERVE -2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:41 INFO routstr.payments RESERVE -2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:41 INFO routstr.payments RESERVE -2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:41 INFO routstr.payments RESERVE -2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:41 INFO routstr.payments RESERVE -2026-09-30 11:32:41 INFO routstr.auth Payment processed successfully -2026-09-30 11:32:41 INFO routstr.payments RESERVE -2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:32:41 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:32:41 INFO routstr.auth Payment settlement finished -2026-09-30 11:32:41 INFO routstr.auth Payment settlement finished -2026-09-30 11:32:42 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:32:42 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:32:42 INFO routstr.auth Calculated token-based cost -2026-09-30 11:32:42 INFO routstr.auth Refunding excess payment -2026-09-30 11:32:42 INFO routstr.auth Refund processed successfully -2026-09-30 11:32:42 INFO routstr.payments FINALIZE -2026-09-30 11:32:42 INFO routstr.auth Payment settlement finished -2026-09-30 11:32:42 INFO routstr.upstream.auto_topup Auto top-up worker started -2026-09-30 11:32:43 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:32:43 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:32:43 ERROR routstr.core.exceptions Unhandled exception -asyncio.exceptions.CancelledError - -The above exception was the direct cause of the following exception: - -TimeoutError -2026-09-30 11:32:43 ERROR uvicorn.error Exception in ASGI application -asyncio.exceptions.CancelledError - -The above exception was the direct cause of the following exception: - -TimeoutError -2026-09-30 11:32:43 INFO routstr.auth Payment settlement finished -2026-09-30 11:32:44 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream -2026-09-30 11:32:44 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:32:44 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:32:44 INFO routstr.auth Calculated token-based cost -2026-09-30 11:32:44 INFO routstr.auth Refunding excess payment -2026-09-30 11:32:44 INFO routstr.auth Refund processed successfully -2026-09-30 11:32:44 INFO routstr.payments FINALIZE -2026-09-30 11:32:44 INFO routstr.auth Payment settlement finished -2026-09-30 11:32:44 ERROR routstr.core.exceptions Unhandled exception -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:32:44 ERROR uvicorn.error Exception in ASGI application -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:32:44 ERROR routstr.upstream.base HTTP request error to upstream -2026-09-30 11:32:44 WARNING routstr.proxy Upstream base failed for model=gpt-4o-mini: Upstream service request timed out -2026-09-30 11:32:52 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:32:55 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:32:55 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:32:55 ERROR uvicorn.error ASGI callable returned without completing response. -2026-09-30 11:32:55 INFO routstr.auth Payment settlement finished diff --git a/repro/router-first.log b/repro/router-first.log deleted file mode 100644 index b8979fc0..00000000 --- a/repro/router-first.log +++ /dev/null @@ -1,115 +0,0 @@ -/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block - return result -2026-09-30 11:28:16 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). -2026-09-30 11:28:16 INFO uvicorn.error Started server process [1] -2026-09-30 11:28:16 INFO uvicorn.error Waiting for application startup. -2026-09-30 11:28:16 INFO routstr.core.main Application startup initiated -2026-09-30 11:28:20 INFO routstr.core.db Database migrations completed successfully -2026-09-30 11:28:21 INFO routstr.core.db Reset reserved balances on startup -2026-09-30 11:28:21 INFO routstr.upstream.helpers Seeding custom provider -2026-09-30 11:28:21 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings -2026-09-30 11:28:22 INFO routstr.proxy Initialized 1 upstream providers -2026-09-30 11:28:22 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider -2026-09-30 11:28:22 INFO routstr.nostr.analytics Usage analytics sharing task started -2026-09-30 11:28:22 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr -2026-09-30 11:28:22 INFO routstr.auth Dead-key pruning disabled (interval <= 0) -2026-09-30 11:28:22 INFO uvicorn.error Application startup complete. -2026-09-30 11:28:22 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18100 (Press CTRL+C to quit) -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:28:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:28:37 INFO routstr.auth Processing payment for request -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:28:37 INFO routstr.payments RESERVE -2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:28:38 INFO routstr.auth Payment settlement finished -2026-09-30 11:28:38 INFO routstr.auth Calculated token-based cost -2026-09-30 11:28:38 INFO routstr.auth Refunding excess payment -2026-09-30 11:28:38 INFO routstr.auth Refund processed successfully -2026-09-30 11:28:38 INFO routstr.payments FINALIZE -2026-09-30 11:28:38 INFO routstr.auth Payment settlement finished -2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:28:38 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:28:38 INFO routstr.auth Calculated token-based cost -2026-09-30 11:28:38 INFO routstr.auth Refunding excess payment -2026-09-30 11:28:38 INFO routstr.auth Refund processed successfully -2026-09-30 11:28:38 INFO routstr.payments FINALIZE -2026-09-30 11:28:38 INFO routstr.auth Payment settlement finished -2026-09-30 11:28:39 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:28:39 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:28:39 INFO routstr.auth Calculated token-based cost -2026-09-30 11:28:39 INFO routstr.auth Finalized payment with additional charge -2026-09-30 11:28:39 INFO routstr.payments FINALIZE -2026-09-30 11:28:39 INFO routstr.auth Payment settlement finished -2026-09-30 11:28:39 ERROR routstr.core.exceptions Unhandled exception -asyncio.exceptions.CancelledError - -The above exception was the direct cause of the following exception: - -TimeoutError -2026-09-30 11:28:39 ERROR uvicorn.error Exception in ASGI application -asyncio.exceptions.CancelledError - -The above exception was the direct cause of the following exception: - -TimeoutError -2026-09-30 11:28:40 ERROR routstr.upstream.base HTTP request error to upstream -2026-09-30 11:28:40 WARNING routstr.proxy Upstream base failed for model=gpt-4o-mini: Upstream service request timed out -2026-09-30 11:28:40 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream -2026-09-30 11:28:40 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:28:40 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:28:40 INFO routstr.auth Calculated token-based cost -2026-09-30 11:28:40 INFO routstr.auth Refunding excess payment -2026-09-30 11:28:40 INFO routstr.auth Refund processed successfully -2026-09-30 11:28:40 INFO routstr.payments FINALIZE -2026-09-30 11:28:40 INFO routstr.auth Payment settlement finished -2026-09-30 11:28:40 ERROR routstr.core.exceptions Unhandled exception -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:28:40 ERROR uvicorn.error Exception in ASGI application -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:28:49 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:28:52 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:28:52 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:28:52 ERROR uvicorn.error ASGI callable returned without completing response. -2026-09-30 11:28:52 INFO routstr.auth Payment settlement finished -2026-09-30 11:28:52 INFO routstr.upstream.auto_topup Auto top-up worker started diff --git a/reservation-repro-main/README.md b/reservation-repro-main/README.md deleted file mode 100644 index e3df04b9..00000000 --- a/reservation-repro-main/README.md +++ /dev/null @@ -1,106 +0,0 @@ -# Current main: real-network streaming reservation reproductions - -## Tested version and environment - -- Commit: `96c8e2f77de8e9f8a0979d17dba0a6d20c78fe89` (local main at investigation time; no remote fetch was performed). -- Unpatched application built using its Dockerfile and frozen lockfile. -- Image: `localhost/routstr-reserved-repro:main`, ID `a787e603f565f3d34e1cc3999793d9dc2d2e3c968eb0ce0ded2f485450719bd0`. -- Podman 5.8.4; Python 3.14; Starlette 1.6.0; Uvicorn 0.31.1. -- Loopback ports 18090 (router), 18091 (dummy upstream), 18092 (diagnostic control). -- Separate container-local SQLite databases and synthetic balances; no original node data or secrets mounted. -- Read timeout accelerated to 3 seconds (confirmed effective); lease expiry to 6 seconds; heartbeat every 2 seconds. Background sweep remains 60 seconds. - -## Results - -| Scenario | Result | -| --- | --- | -| Finite stream with usage and DONE | Charged normally, zero reservation | -| One chunk then silence, client connected | Read timeout fired, estimated usage charged, zero reservation | -| One chunk then silence, client disconnected after 1 second | Reservation cleared on upstream read timeout; prompt disconnect cleanup was not demonstrated | -| No upstream response headers | Timeout produced HTTP 424; reservation released | -| Endless content stream, client disconnected after 1 second | Continued renewing; exact refund HTTP 400 persisted across background sweep | -| SSE comment-only keepalives every 0.5 seconds | No meaningful content or completion, but lease renewed and refund blocked; remained active after client disconnected | -| Flood stream to client that never reads | Lease renewed while client was stalled; still renewed after client socket closed | - -The three problematic streams retained 11-msat reservations through the full observation window. They began at timestamp 1790766366; at 1790766635, all remained active with lease timestamps 1790766634. Thus renewal continued for roughly 269 seconds, far beyond the 3-second read timeout, 6-second lease timeout, and multiple 60-second sweep intervals. All test clients were gone by approximately 1790766439. - -This proves persistence for minutes, not a measured days-long run. No new inference requests were made for the keys during observation; refund probes did not renew the leases. - -The flood scenario sends 64-KiB content deltas rapidly and uses a 1-KiB client receive buffer. It exercises a real non-reading downstream socket, but no live task-stack capture was collected to establish the precise blocked await at each snapshot. - -## Why the newer timeout is insufficient - -The read timeout is an inactivity timeout for upstream reads. Endless content or SSE keepalive bytes avoid it. A downstream-send wait is not bounded by it. - -More importantly, the runtime did not reliably propagate downstream disconnect into termination of these streams. Closed clients left upstream connections established and reservation owners alive, so heartbeats kept making the durable rows fresh. The sweeper therefore correctly declined to release them under its current policy. - -## Framework evidence and diagnostic control - -Captured sources (`starlette-source.txt`, `uvicorn-source.txt`) show: - -- Uvicorn 0.31.1's httptools protocol advertises ASGI HTTP spec 2.4. -- Its `send()` returns silently when `self.disconnected` is true; it does not raise an OSError. -- Starlette's StreamingResponse for ASGI >=2.4 relies on a send OSError to signal client disconnect, rather than running its older explicit disconnect listener. -- BaseHTTPMiddleware's outer streaming wrapper also does not explicitly listen for disconnect. - -This is a concrete framework compatibility concern consistent with the observations. Deterministic confirmation via a server-version/spec comparison or task instrumentation remains future work. - -A diagnostic second router removed only LoggingMiddleware using `no_logging_app.py`. Endless and keepalive clients still left active reservations after disconnect (`control-results.txt`). Thus LoggingMiddleware alone is not sufficient to explain the disconnect leak in this environment. This control is not a proposed production patch. - -When the dummy upstream was forcibly stopped, the control router finalized both streams. The unmodified main router still showed the three reservations active five seconds afterward and subsequently needed SIGKILL after a ten-second shutdown grace period. Logs showed upstream termination warnings but no completed settlement for those three in the captured window. The exact finalization blockage was not traced; it should be investigated separately, potentially including middleware delivery/backpressure interactions. Do not assert that upstream termination always clears these main reservations. - -## Reproduce - -From project root: - -```bash -podman build --build-arg GIT_COMMIT=$(git rev-parse HEAD) --build-arg GIT_TAG=main \ - -t localhost/routstr-reserved-repro:main . - -podman run -d --name reserved-dummy-main --network host \ - -v "$PWD/reservation-repro-main:/repro:ro,Z" \ - --entrypoint /.venv/bin/python localhost/routstr-reserved-repro:main \ - -m uvicorn dummy_upstream:app --app-dir /repro --host 127.0.0.1 --port 18091 - -podman run -d --name reserved-router-main --network host \ - -e DATABASE_URL=sqlite+aiosqlite:////tmp/reserved-main.db \ - -e UPSTREAM_BASE_URL=http://127.0.0.1:18091/v1 -e UPSTREAM_API_KEY=dummy \ - -e STALE_RESERVATION_TIMEOUT_SECONDS=6 -e UPSTREAM_READ_TIMEOUT=3 \ - -e CASHU_MINTS= -e ENABLE_PRICING_REFRESH=false \ - -e MODELS_REFRESH_INTERVAL_SECONDS=0 -e ADMIN_PASSWORD=local-repro-only \ - --entrypoint /.venv/bin/python localhost/routstr-reserved-repro:main \ - -m uvicorn routstr.core.main:app --host 127.0.0.1 --port 18090 -``` - -Wait for application startup and verify `/v1/models` includes gpt-4o-mini. Model/pricing discovery uses external services; this is not fully offline. - -```bash -podman exec -i reserved-router-main /.venv/bin/python - <<'PY' -import asyncio -from routstr.core.db import ApiKey, create_session -async def main(): - async with create_session() as s: - for k in ['finite','silent','silent-disconnect','endless-disconnect','keepalive','flood','header']: - s.add(ApiKey(hashed_key='main-'+k, balance=1000000000)) - await s.commit() -asyncio.run(main()) -PY - -.venv/bin/python reservation-repro-main/probe.py -``` - -The probe runs approximately 80 seconds, snapshots the DB, attempts refunds only on reserved keys (not actual Cashu payouts), and closes all clients. Later DB snapshots show continued renewal. Use fresh container names/databases on repeats or deliberately remove only the retained reproduction containers first. Do not overwrite original node containers. - -## Evidence and remaining work - -- `results.txt`: scenario matrix snapshots and refund errors. -- `connections.txt`: upstream sockets remained after downstream sockets disappeared. -- `final-before-stop.json`: continued renewal roughly 269 seconds after start. -- `after-upstream-stop.json`: reservations still active in unmodified main five seconds after upstream termination. -- `router.log`, `upstream.log`: application evidence before router shutdown. -- `control-results.txt`, `control-router.log`: comparison without LoggingMiddleware. -- `starlette-source.txt`, `uvicorn-source.txt`: installed framework behavior. - -Need: real-network regression tests, framework compatibility correction/verification, explicit disconnect monitoring that reaches upstream ownership, bounded downstream delivery, total request lifetime, and task-stack diagnostics for finalization stalls. Database fault injection, restart/multi-worker behavior, and alternate API routes were not tested here. - -All three main reproduction containers were stopped. The unmodified main router required SIGKILL; its retained database may contain active reservations. No application source fixes were made. diff --git a/reservation-repro-main/after-upstream-stop.json b/reservation-repro-main/after-upstream-stop.json deleted file mode 100644 index a4c70696..00000000 --- a/reservation-repro-main/after-upstream-stop.json +++ /dev/null @@ -1 +0,0 @@ -{"time": 1790766644.5905168, "keys": [["main-finite", 0], ["main-silent", 0], ["main-silent-disconnect", 0], ["main-endless-disconnect", 11], ["main-keepalive", 11], ["main-flood", 11], ["main-header", 0]], "rows": [["main-flood", "active", 1790766636], ["main-keepalive", "active", 1790766636], ["main-endless-disconnect", "active", 1790766636], ["main-silent", "charged", 1790766368], ["main-silent-disconnect", "charged", 1790766368], ["main-header", "released", 1790766368], ["main-finite", "charged", 1790766366]]} diff --git a/reservation-repro-main/connections.txt b/reservation-repro-main/connections.txt deleted file mode 100644 index 2911cbec..00000000 --- a/reservation-repro-main/connections.txt +++ /dev/null @@ -1,7 +0,0 @@ -ESTAB 0 0 127.0.0.1:18091 127.0.0.1:36380 users:(("python",pid=1480010,fd=7)) -ESTAB 0 0 127.0.0.1:36380 127.0.0.1:18091 users:(("python",pid=1480035,fd=31)) -ESTAB 0 0 127.0.0.1:36394 127.0.0.1:18091 users:(("python",pid=1480035,fd=32)) -ESTAB 0 0 127.0.0.1:36402 127.0.0.1:18091 users:(("python",pid=1480035,fd=33)) -CLOSE-WAIT 1 0 127.0.0.1:36456 127.0.0.1:18091 users:(("python",pid=1480035,fd=37)) -ESTAB 0 188 127.0.0.1:18091 127.0.0.1:36402 users:(("python",pid=1480010,fd=9)) -ESTAB 0 0 127.0.0.1:18091 127.0.0.1:36394 users:(("python",pid=1480010,fd=8)) diff --git a/reservation-repro-main/control-results.txt b/reservation-repro-main/control-results.txt deleted file mode 100644 index 84aee312..00000000 --- a/reservation-repro-main/control-results.txt +++ /dev/null @@ -1,5 +0,0 @@ -endless-control 200 -keepalive-control 200 -[('endless-control', 12), ('keepalive-control', 12)] -[('endless-control', 'active'), ('keepalive-control', 'active')] - diff --git a/reservation-repro-main/control-router.log b/reservation-repro-main/control-router.log deleted file mode 100644 index de3a8147..00000000 --- a/reservation-repro-main/control-router.log +++ /dev/null @@ -1,43 +0,0 @@ -/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block - return result -2026-09-30 11:08:57 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). -2026-09-30 11:08:57 INFO uvicorn.error Started server process [1] -2026-09-30 11:08:57 INFO uvicorn.error Waiting for application startup. -2026-09-30 11:08:57 INFO routstr.core.main Application startup initiated -2026-09-30 11:08:59 INFO routstr.core.db Database migrations completed successfully -2026-09-30 11:08:59 INFO routstr.core.db Reset reserved balances on startup -2026-09-30 11:08:59 INFO routstr.upstream.helpers Seeding custom provider -2026-09-30 11:08:59 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings -2026-09-30 11:09:00 INFO routstr.proxy Initialized 1 upstream providers -2026-09-30 11:09:00 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider -2026-09-30 11:09:00 INFO routstr.nostr.analytics Usage analytics sharing task started -2026-09-30 11:09:00 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr -2026-09-30 11:09:00 INFO routstr.auth Dead-key pruning disabled (interval <= 0) -2026-09-30 11:09:00 INFO uvicorn.error Application startup complete. -2026-09-30 11:09:00 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18092 (Press CTRL+C to quit) -2026-09-30 11:09:30 INFO routstr.upstream.auto_topup Auto top-up worker started -2026-09-30 11:09:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:09:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:09:37 INFO routstr.auth Processing payment for request -2026-09-30 11:09:37 INFO routstr.auth Existing sk- API key found -2026-09-30 11:09:37 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:09:37 INFO routstr.auth Processing payment for request -2026-09-30 11:09:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:09:37 INFO routstr.payments RESERVE -2026-09-30 11:09:37 INFO routstr.auth Payment processed successfully -2026-09-30 11:09:37 INFO routstr.payments RESERVE -2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete -2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete -2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:10:38 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:10:38 INFO routstr.auth Calculated token-based cost -2026-09-30 11:10:38 INFO routstr.auth Finalized payment with additional charge -2026-09-30 11:10:38 INFO routstr.payments FINALIZE -2026-09-30 11:10:38 INFO routstr.auth Payment settlement finished -2026-09-30 11:10:38 INFO routstr.auth Calculated token-based cost -2026-09-30 11:10:38 INFO routstr.auth Refunding excess payment -2026-09-30 11:10:38 INFO routstr.auth Refund processed successfully -2026-09-30 11:10:38 INFO routstr.payments FINALIZE -2026-09-30 11:10:38 INFO routstr.auth Payment settlement finished diff --git a/reservation-repro-main/dummy_upstream.py b/reservation-repro-main/dummy_upstream.py deleted file mode 100644 index 1be75ecc..00000000 --- a/reservation-repro-main/dummy_upstream.py +++ /dev/null @@ -1,45 +0,0 @@ -"""Loopback-only streaming fixture; no router monkeypatches.""" -import asyncio -import json -import time -from fastapi import FastAPI, Request -from fastapi.responses import StreamingResponse - -app = FastAPI() -events = [] - -@app.get('/events') -async def history(): - return events - -@app.get('/v1/models') -async def models(): - return {'object': 'list', 'data': [{'id': 'gpt-4o-mini', 'object': 'model', 'created': 1, 'owned_by': 'repro'}]} - -@app.post('/v1/chat/completions') -async def completions(request: Request): - body = await request.json() - mode = body.get('messages', [{}])[0].get('content', 'finite') - events.append({'event': 'start', 'mode': mode, 'time': time.time()}) - if mode.startswith('header'): - await asyncio.sleep(3600) - async def stream(): - count = 0 - try: - while True: - if mode.startswith('keepalive'): - yield ': ping\n\n' - else: - chunk = {'id': 'repro', 'object': 'chat.completion.chunk', 'created': int(time.time()), 'model': 'gpt-4o-mini', 'choices': [{'index': 0, 'delta': {'content': 'x' * (65536 if mode.startswith('flood') else 1)}, 'finish_reason': None}]} - yield 'data: ' + json.dumps(chunk) + '\n\n' - count += 1 - if mode == 'finite' and count >= 3: - yield 'data: ' + json.dumps({'id': 'repro', 'object': 'chat.completion.chunk', 'model': 'gpt-4o-mini', 'choices': [], 'usage': {'prompt_tokens': 1, 'completion_tokens': count, 'total_tokens': count + 1}}) + '\n\n' - yield 'data: [DONE]\n\n' - return - await asyncio.sleep(3600 if mode.startswith('silent') else (0.001 if mode.startswith('flood') else 0.5)) - finally: - event = {'event': 'close', 'mode': mode, 'chunks': count, 'time': time.time()} - events.append(event) - print(json.dumps(event), flush=True) - return StreamingResponse(stream(), media_type='text/event-stream') diff --git a/reservation-repro-main/final-before-stop.json b/reservation-repro-main/final-before-stop.json deleted file mode 100644 index 4647ae85..00000000 --- a/reservation-repro-main/final-before-stop.json +++ /dev/null @@ -1 +0,0 @@ -{"time": 1790766635.1679196, "keys": [["main-finite", 0], ["main-silent", 0], ["main-silent-disconnect", 0], ["main-endless-disconnect", 11], ["main-keepalive", 11], ["main-flood", 11], ["main-header", 0]], "rows": [["main-flood", "active", 1790766634], ["main-keepalive", "active", 1790766634], ["main-endless-disconnect", "active", 1790766634], ["main-silent", "charged", 1790766368], ["main-silent-disconnect", "charged", 1790766368], ["main-header", "released", 1790766368], ["main-finite", "charged", 1790766366]]} diff --git a/reservation-repro-main/no_logging_app.py b/reservation-repro-main/no_logging_app.py deleted file mode 100644 index 16c4cdfc..00000000 --- a/reservation-repro-main/no_logging_app.py +++ /dev/null @@ -1,4 +0,0 @@ -"""Diagnostic comparison ONLY: remove LoggingMiddleware from unchanged image app.""" -from routstr.core.main import app -from routstr.core.middleware import LoggingMiddleware -app.user_middleware = [m for m in app.user_middleware if m.cls is not LoggingMiddleware] diff --git a/reservation-repro-main/probe.py b/reservation-repro-main/probe.py deleted file mode 100644 index 57653e5e..00000000 --- a/reservation-repro-main/probe.py +++ /dev/null @@ -1,60 +0,0 @@ -import asyncio -import json -import socket -import subprocess -import time -import httpx - -BASE='http://127.0.0.1:18090' - -def snapshot(): - code="import sqlite3,json,time; c=sqlite3.connect('/tmp/reserved-main.db'); c.row_factory=sqlite3.Row; print(json.dumps({'time':time.time(),'keys':[dict(r) for r in c.execute(\"select hashed_key,balance,reserved_balance,reserved_at from api_keys where hashed_key like 'main-%'\")],'rows':[dict(r) for r in c.execute(\"select * from reservation_releases where key_hash like 'main-%'\")]}))" - return json.loads(subprocess.check_output(['podman','exec','reserved-router-main','/.venv/bin/python','-c',code],text=True)) - -async def consume(mode): - try: - async with httpx.AsyncClient(timeout=None) as c: - async with c.stream('POST',BASE+'/v1/chat/completions',headers={'Authorization':'Bearer sk-main-'+mode},json={'model':'gpt-4o-mini','messages':[{'role':'user','content':mode}],'stream':True,'max_tokens':10}) as r: - print('STREAM',mode,r.status_code,flush=True) - async for _ in r.aiter_bytes(): pass - print('ENDED',mode,flush=True) - except asyncio.CancelledError: - print('CLIENT_DISCONNECTED',mode,flush=True) - raise - except Exception as e: - print('CLIENT_ERROR',mode,type(e).__name__,str(e),flush=True) - -async def report(label): - print(label,json.dumps(snapshot()),flush=True) - async with httpx.AsyncClient(timeout=5) as c: - for mode in ['silent-disconnect','endless-disconnect','keepalive','flood','header']: - # Only attempt payout while reserved: avoid requiring a real mint. - if next(k for k in snapshot()['keys'] if k['hashed_key']=='main-'+mode)['reserved_balance']: - r=await c.post(BASE+'/v1/wallet/refund',headers={'Authorization':'Bearer sk-main-'+mode}) - print('REFUND',mode,r.status_code,r.text,flush=True) - print('UPSTREAM_EVENTS',json.dumps((await c.get('http://127.0.0.1:18091/events')).json()),flush=True) - -async def main(): - modes=['finite','silent','silent-disconnect','endless-disconnect','keepalive','header'] - tasks={m:asyncio.create_task(consume(m)) for m in modes} - # Real client with a small receive buffer, never draining the HTTP response. - sock=socket.socket(); sock.setsockopt(socket.SOL_SOCKET,socket.SO_RCVBUF,1024); sock.connect(('127.0.0.1',18090)) - body=json.dumps({'model':'gpt-4o-mini','messages':[{'role':'user','content':'flood'}],'stream':True,'max_tokens':10}).encode() - sock.sendall(b'POST /v1/chat/completions HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer sk-main-flood\r\nContent-Type: application/json\r\nContent-Length: '+str(len(body)).encode()+b'\r\n\r\n'+body) - await asyncio.sleep(1) - for m in ['silent-disconnect','endless-disconnect']: - tasks[m].cancel() - await asyncio.gather(tasks['silent-disconnect'],tasks['endless-disconnect'],return_exceptions=True) - await asyncio.sleep(9) - await report('AT_10_SECONDS') - await asyncio.sleep(60) - await report('AFTER_SWEEP') - sock.close() - tasks['keepalive'].cancel() - await asyncio.gather(tasks['keepalive'],return_exceptions=True) - await asyncio.sleep(8) - await report('AFTER_ALL_CLIENTS_CLOSED') - for task in tasks.values(): task.cancel() - await asyncio.gather(*tasks.values(),return_exceptions=True) - -asyncio.run(main()) diff --git a/reservation-repro-main/results.txt b/reservation-repro-main/results.txt deleted file mode 100644 index 7f21d733..00000000 --- a/reservation-repro-main/results.txt +++ /dev/null @@ -1,27 +0,0 @@ -STREAM keepalive 200 -STREAM endless-disconnect 200 -STREAM silent 200 -STREAM silent-disconnect 200 -STREAM finite 200 -CLIENT_DISCONNECTED silent-disconnect -CLIENT_DISCONNECTED endless-disconnect -ENDED finite -ENDED silent -STREAM header 424 -ENDED header -AT_10_SECONDS {"time": 1790766376.4572322, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "50569f573caf4d6fb7916da3570493db", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 11, "status": "active", "created_at": 1790766376}, {"id": "0ffb4d61dc0d4aaf9e518c74b7afd1bb", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 11, "status": "active", "created_at": 1790766376}, {"id": "66d5bdd9d9814f6fb1576ed6708f431e", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 11, "status": "active", "created_at": 1790766374}, {"id": "a891d80b8db64e488f8896936cd5f2fe", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "6ebb7f0829bd4f569d9d2516dabd9fed", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "e14358bc6e7249c0ac7335c9d87e7b43", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 11, "status": "released", "created_at": 1790766368}, {"id": "26c897009e294f94b75ca51f071d335a", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 11, "status": "charged", "created_at": 1790766366}]} -REFUND endless-disconnect 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"c272c298-bace-482c-926f-0c56fdaeaa5e"} -REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"3c1124f6-1c21-4417-b4fb-2ffdead58c31"} -REFUND flood 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"afca3cab-361a-48f6-85e9-58247c17a5f2"} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790766366.393702}, {"event": "start", "mode": "keepalive", "time": 1790766366.408879}, {"event": "start", "mode": "endless-disconnect", "time": 1790766366.4305305}, {"event": "start", "mode": "silent", "time": 1790766366.4564564}, {"event": "start", "mode": "silent-disconnect", "time": 1790766366.4789124}, {"event": "start", "mode": "header", "time": 1790766366.5032742}, {"event": "start", "mode": "finite", "time": 1790766366.5216281}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321}] -AFTER_SWEEP {"time": 1790766437.9892845, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "50569f573caf4d6fb7916da3570493db", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 11, "status": "active", "created_at": 1790766436}, {"id": "0ffb4d61dc0d4aaf9e518c74b7afd1bb", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 11, "status": "active", "created_at": 1790766436}, {"id": "66d5bdd9d9814f6fb1576ed6708f431e", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 11, "status": "active", "created_at": 1790766436}, {"id": "a891d80b8db64e488f8896936cd5f2fe", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "6ebb7f0829bd4f569d9d2516dabd9fed", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "e14358bc6e7249c0ac7335c9d87e7b43", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 11, "status": "released", "created_at": 1790766368}, {"id": "26c897009e294f94b75ca51f071d335a", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 11, "status": "charged", "created_at": 1790766366}]} -REFUND endless-disconnect 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"4a00cdce-d049-4be8-940f-2a652349c1f9"} -REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"bcf3b7d4-ae59-494c-8049-bb43476211e8"} -REFUND flood 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"4b459580-84d1-480a-b1cb-28898a04395e"} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790766366.393702}, {"event": "start", "mode": "keepalive", "time": 1790766366.408879}, {"event": "start", "mode": "endless-disconnect", "time": 1790766366.4305305}, {"event": "start", "mode": "silent", "time": 1790766366.4564564}, {"event": "start", "mode": "silent-disconnect", "time": 1790766366.4789124}, {"event": "start", "mode": "header", "time": 1790766366.5032742}, {"event": "start", "mode": "finite", "time": 1790766366.5216281}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321}] -CLIENT_DISCONNECTED keepalive -AFTER_ALL_CLIENTS_CLOSED {"time": 1790766447.4688976, "keys": [{"hashed_key": "main-finite", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-silent-disconnect", "balance": 999999997, "reserved_balance": 0, "reserved_at": null}, {"hashed_key": "main-endless-disconnect", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-keepalive", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-flood", "balance": 1000000000, "reserved_balance": 11, "reserved_at": 1790766366}, {"hashed_key": "main-header", "balance": 1000000000, "reserved_balance": 0, "reserved_at": null}], "rows": [{"id": "50569f573caf4d6fb7916da3570493db", "key_hash": "main-flood", "billing_key_hash": "main-flood", "reserved_msats": 11, "status": "active", "created_at": 1790766447}, {"id": "0ffb4d61dc0d4aaf9e518c74b7afd1bb", "key_hash": "main-keepalive", "billing_key_hash": "main-keepalive", "reserved_msats": 11, "status": "active", "created_at": 1790766446}, {"id": "66d5bdd9d9814f6fb1576ed6708f431e", "key_hash": "main-endless-disconnect", "billing_key_hash": "main-endless-disconnect", "reserved_msats": 11, "status": "active", "created_at": 1790766446}, {"id": "a891d80b8db64e488f8896936cd5f2fe", "key_hash": "main-silent", "billing_key_hash": "main-silent", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "6ebb7f0829bd4f569d9d2516dabd9fed", "key_hash": "main-silent-disconnect", "billing_key_hash": "main-silent-disconnect", "reserved_msats": 11, "status": "charged", "created_at": 1790766368}, {"id": "e14358bc6e7249c0ac7335c9d87e7b43", "key_hash": "main-header", "billing_key_hash": "main-header", "reserved_msats": 11, "status": "released", "created_at": 1790766368}, {"id": "26c897009e294f94b75ca51f071d335a", "key_hash": "main-finite", "billing_key_hash": "main-finite", "reserved_msats": 11, "status": "charged", "created_at": 1790766366}]} -REFUND endless-disconnect 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"67367fbb-2fd6-4ce7-a0e3-3b1d0f30cb76"} -REFUND keepalive 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"96f8b70c-584e-476d-aa9c-ea9092e9393f"} -REFUND flood 400 {"detail":"Cannot refund key. There are ongoing requests for this api key.","request_id":"60ac9762-dc92-4e08-9456-77346a70a63d"} -UPSTREAM_EVENTS [{"event": "start", "mode": "flood", "time": 1790766366.393702}, {"event": "start", "mode": "keepalive", "time": 1790766366.408879}, {"event": "start", "mode": "endless-disconnect", "time": 1790766366.4305305}, {"event": "start", "mode": "silent", "time": 1790766366.4564564}, {"event": "start", "mode": "silent-disconnect", "time": 1790766366.4789124}, {"event": "start", "mode": "header", "time": 1790766366.5032742}, {"event": "start", "mode": "finite", "time": 1790766366.5216281}, {"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321}] diff --git a/reservation-repro-main/router.log b/reservation-repro-main/router.log deleted file mode 100644 index f2e7c738..00000000 --- a/reservation-repro-main/router.log +++ /dev/null @@ -1,114 +0,0 @@ -/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block - return result -2026-09-30 11:05:28 WARNING routstr.core.main UI dist directory not found at /app/ui_out; serving API only. Run `make ui-build` to build the static UI served from here, or `make ui-dev` for the Next.js dev server with hot reload on :3000 (it targets this backend on :8000). -2026-09-30 11:05:28 INFO uvicorn.error Started server process [1] -2026-09-30 11:05:28 INFO uvicorn.error Waiting for application startup. -2026-09-30 11:05:28 INFO routstr.core.main Application startup initiated -2026-09-30 11:05:30 INFO routstr.core.db Database migrations completed successfully -2026-09-30 11:05:30 INFO routstr.core.db Reset reserved balances on startup -2026-09-30 11:05:30 INFO routstr.upstream.helpers Seeding custom provider -2026-09-30 11:05:30 INFO routstr.upstream.helpers Seeded 1 upstream providers from settings -2026-09-30 11:05:31 INFO routstr.proxy Initialized 1 upstream providers -2026-09-30 11:05:31 INFO routstr.nostr.listing Nostr private key not configured (NSEC); waiting for one to be set before announcing this provider -2026-09-30 11:05:31 INFO routstr.nostr.analytics Usage analytics sharing task started -2026-09-30 11:05:31 INFO routstr.nostr.analytics NSEC is not configured; skipping analytics sharing to Nostr -2026-09-30 11:05:31 INFO routstr.auth Dead-key pruning disabled (interval <= 0) -2026-09-30 11:05:31 INFO uvicorn.error Application startup complete. -2026-09-30 11:05:31 INFO uvicorn.error Uvicorn running on http://127.0.0.1:18090 (Press CTRL+C to quit) -2026-09-30 11:06:01 INFO routstr.upstream.auto_topup Auto top-up worker started -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Existing sk- API key found -2026-09-30 11:06:06 INFO routstr.proxy Bearer token validated successfully -2026-09-30 11:06:06 INFO routstr.auth Processing payment for request -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:06 INFO routstr.auth Payment processed successfully -2026-09-30 11:06:06 INFO routstr.payments RESERVE -2026-09-30 11:06:07 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:06:07 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:06:07 INFO routstr.auth Calculated token-based cost -2026-09-30 11:06:07 INFO routstr.auth Refunding excess payment -2026-09-30 11:06:07 INFO routstr.auth Refund processed successfully -2026-09-30 11:06:07 INFO routstr.payments FINALIZE -2026-09-30 11:06:07 INFO routstr.auth Payment settlement finished -2026-09-30 11:06:09 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream -2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:06:09 INFO routstr.auth Calculated token-based cost -2026-09-30 11:06:09 INFO routstr.auth Refunding excess payment -2026-09-30 11:06:09 WARNING routstr.upstream.base Streaming interrupted; finalizing before closing upstream -2026-09-30 11:06:09 INFO routstr.auth Refund processed successfully -2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Applied model-specific pricing -2026-09-30 11:06:09 INFO routstr.payment.cost_calculation Calculated token-based cost -2026-09-30 11:06:09 INFO routstr.auth Calculated token-based cost -2026-09-30 11:06:09 INFO routstr.auth Refunding excess payment -2026-09-30 11:06:09 ERROR routstr.upstream.base HTTP request error to upstream -2026-09-30 11:06:09 WARNING routstr.proxy Upstream base failed for model=gpt-4o-mini: Upstream service request timed out -2026-09-30 11:06:09 INFO routstr.auth Refund processed successfully -2026-09-30 11:06:09 INFO routstr.payments FINALIZE -2026-09-30 11:06:09 INFO routstr.auth Payment settlement finished -2026-09-30 11:06:09 ERROR routstr.core.exceptions Unhandled exception -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:06:09 ERROR uvicorn.error Exception in ASGI application -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:06:09 INFO routstr.payments FINALIZE -2026-09-30 11:06:09 INFO routstr.auth Payment settlement finished -2026-09-30 11:06:09 ERROR routstr.core.exceptions Unhandled exception -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:06:09 ERROR uvicorn.error Exception in ASGI application -httpcore.ReadTimeout - -The above exception was the direct cause of the following exception: - -httpx.ReadTimeout -2026-09-30 11:06:16 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:06:17 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:06:17 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:07:18 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:07:18 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:07:19 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:07:28 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:07:28 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:07:28 INFO routstr.core.exceptions HTTP 400 on /v1/wallet/refund: Cannot refund key. There are ongoing requests for this api key. -2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete -2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete -2026-09-30 11:10:38 WARNING routstr.upstream.base Upstream stream ended before the response was complete diff --git a/reservation-repro-main/starlette-source.txt b/reservation-repro-main/starlette-source.txt deleted file mode 100644 index 78736b3f..00000000 --- a/reservation-repro-main/starlette-source.txt +++ /dev/null @@ -1,169 +0,0 @@ - async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: - if scope["type"] != "http": - await self.app(scope, receive, send) - return - - request = _CachedRequest(scope, receive) - wrapped_receive = request.wrapped_receive - response_sent = anyio.Event() - app_exc: Exception | None = None - exception_already_raised = False - - async def call_next(request: Request) -> Response: - async def receive_or_disconnect() -> Message: - if response_sent.is_set(): - return {"type": "http.disconnect"} - - async with anyio.create_task_group() as task_group: - - async def wrap(func: Callable[[], Awaitable[T]]) -> T: - result = await func() - task_group.cancel_scope.cancel() - return result - - task_group.start_soon(wrap, response_sent.wait) - message = await wrap(wrapped_receive) - - if response_sent.is_set(): - return {"type": "http.disconnect"} - - return message - - async def send_no_error(message: Message) -> None: - try: - await send_stream.send(message) - except anyio.BrokenResourceError: - # recv_stream has been closed, i.e. response_sent has been set. - return - - async def coro() -> None: - nonlocal app_exc - - with send_stream: - try: - await self.app(scope, receive_or_disconnect, send_no_error) - except Exception as exc: - app_exc = exc - - task_group.start_soon(coro) - - try: - message = await recv_stream.receive() - info = message.get("info", None) - if message["type"] == "http.response.debug" and info is not None: - message = await recv_stream.receive() - except anyio.EndOfStream: - if app_exc is not None: - nonlocal exception_already_raised - exception_already_raised = True - # Prevent `anyio.EndOfStream` from polluting app exception context. - # If both cause and context are None then the context is suppressed - # and `anyio.EndOfStream` is not present in the exception traceback. - # If exception cause is not None then it is propagated with - # reraising here. - # If exception has no cause but has context set then the context is - # propagated as a cause with the reraise. This is necessary in order - # to prevent `anyio.EndOfStream` from polluting the exception - # context. - raise app_exc from app_exc.__cause__ or app_exc.__context__ - raise RuntimeError("No response returned.") - - assert message["type"] == "http.response.start" - - async def body_stream() -> BodyStreamGenerator: - async for message in recv_stream: - if message["type"] == "http.response.pathsend": - yield message - break - assert message["type"] == "http.response.body", f"Unexpected message: {message}" - body = message.get("body", b"") - if body: - yield body - if not message.get("more_body", False): - break - - response = _StreamingResponse(status_code=message["status"], content=body_stream(), info=info) - response.raw_headers = message["headers"] - return response - - streams: anyio.create_memory_object_stream[Message] = anyio.create_memory_object_stream() - send_stream, recv_stream = streams - with recv_stream, send_stream: - async with create_collapsing_task_group() as task_group: - response = await self.dispatch_func(request, call_next) - await response(scope, wrapped_receive, send) - response_sent.set() - recv_stream.close() - if app_exc is not None and not exception_already_raised: - raise app_exc - -class _StreamingResponse(Response): - def __init__( - self, - content: AsyncContentStream, - status_code: int = 200, - headers: Mapping[str, str] | None = None, - media_type: str | None = None, - info: Mapping[str, Any] | None = None, - ) -> None: - self.info = info - self.body_iterator = content - self.status_code = status_code - self.media_type = media_type - self.init_headers(headers) - self.background = None - - async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: - if self.info is not None: - await send({"type": "http.response.debug", "info": self.info}) - await send( - { - "type": "http.response.start", - "status": self.status_code, - "headers": self.raw_headers, - } - ) - - should_close_body = True - async for chunk in self.body_iterator: - if isinstance(chunk, dict): - # We got an ASGI message which is not response body (eg: pathsend) - should_close_body = False - await send(chunk) - continue - await send({"type": "http.response.body", "body": chunk, "more_body": True}) - - if should_close_body: - await send({"type": "http.response.body", "body": b"", "more_body": False}) - - if self.background: - await self.background() - - async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: - if scope["type"] == "websocket": - send = self._wrap_websocket_denial_send(send) - await self.stream_response(send) - if self.background is not None: - await self.background() - return - - spec_version = tuple(map(int, scope.get("asgi", {}).get("spec_version", "2.0").split("."))) - - if spec_version >= (2, 4): - try: - await self.stream_response(send) - except OSError: - raise ClientDisconnect() - else: - async with create_collapsing_task_group() as task_group: - - async def wrap(func: Callable[[], Awaitable[None]]) -> None: - await func() - task_group.cancel_scope.cancel() - - task_group.start_soon(wrap, partial(self.stream_response, send)) - await wrap(partial(self.listen_for_disconnect, receive)) - - if self.background is not None: - await self.background() - diff --git a/reservation-repro-main/upstream.log b/reservation-repro-main/upstream.log deleted file mode 100644 index 2f124bcd..00000000 --- a/reservation-repro-main/upstream.log +++ /dev/null @@ -1,23 +0,0 @@ -/.venv/lib/python3.14/site-packages/anyio/from_thread.py:119: SyntaxWarning: 'return' in a 'finally' block - return result -INFO: Started server process [1] -INFO: Waiting for application startup. -INFO: Application startup complete. -INFO: Uvicorn running on http://127.0.0.1:18091 (Press CTRL+C to quit) -INFO: 127.0.0.1:59686 - "GET /v1/models HTTP/1.1" 200 OK -INFO: 127.0.0.1:36380 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:36394 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:36402 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:36418 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:36434 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:36456 - "POST /v1/chat/completions HTTP/1.1" 200 OK -{"event": "close", "mode": "finite", "chunks": 3, "time": 1790766367.5249321} -INFO: 127.0.0.1:54322 - "GET /events HTTP/1.1" 200 OK -INFO: 127.0.0.1:51770 - "GET /events HTTP/1.1" 200 OK -INFO: 127.0.0.1:42140 - "GET /events HTTP/1.1" 200 OK -INFO: 127.0.0.1:50608 - "GET /v1/models HTTP/1.1" 200 OK -INFO: 127.0.0.1:46164 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:46178 - "POST /v1/chat/completions HTTP/1.1" 200 OK -INFO: 127.0.0.1:39728 - "GET /events HTTP/1.1" 200 OK -INFO: Shutting down -INFO: Waiting for connections to close. (CTRL+C to force quit) diff --git a/reservation-repro-main/uvicorn-source.txt b/reservation-repro-main/uvicorn-source.txt deleted file mode 100644 index 2c5d74b3..00000000 --- a/reservation-repro-main/uvicorn-source.txt +++ /dev/null @@ -1,125 +0,0 @@ - async def send(self, message: ASGISendEvent) -> None: - message_type = message["type"] - - if self.flow.write_paused and not self.disconnected: - await self.flow.drain() # pragma: full coverage - - if self.disconnected: - return # pragma: full coverage - - if not self.response_started: - # Sending response status line and headers - if message_type != "http.response.start": - msg = "Expected ASGI message 'http.response.start', but got '%s'." - raise RuntimeError(msg % message_type) - message = cast("HTTPResponseStartEvent", message) - - self.response_started = True - self.waiting_for_100_continue = False - - status_code = message["status"] - headers = self.default_headers + list(message.get("headers", [])) - - if CLOSE_HEADER in self.scope["headers"] and CLOSE_HEADER not in headers: - headers = headers + [CLOSE_HEADER] - - if self.access_log: - self.access_logger.info( - '%s - "%s %s HTTP/%s" %d', - get_client_addr(self.scope), - self.scope["method"], - get_path_with_query_string(self.scope), - self.scope["http_version"], - status_code, - ) - - # Write response status line and headers - content = [STATUS_LINE[status_code]] - - for name, value in headers: - if HEADER_RE.search(name): - raise RuntimeError("Invalid HTTP header name.") # pragma: full coverage - if HEADER_VALUE_RE.search(value): - raise RuntimeError("Invalid HTTP header value.") - - name = name.lower() - if name == b"content-length" and self.chunked_encoding is None: - self.expected_content_length = int(value.decode()) - self.chunked_encoding = False - elif name == b"transfer-encoding" and value.lower() == b"chunked": - self.expected_content_length = 0 - self.chunked_encoding = True - elif name == b"connection" and value.lower() == b"close": - self.keep_alive = False - content.extend([name, b": ", value, b"\r\n"]) - - if self.chunked_encoding is None and self.scope["method"] != "HEAD" and status_code not in (204, 304): - # Neither content-length nor transfer-encoding specified - self.chunked_encoding = True - content.append(b"transfer-encoding: chunked\r\n") - - content.append(b"\r\n") - self.transport.write(b"".join(content)) - - elif not self.response_complete: - # Sending response body - if message_type != "http.response.body": - msg = "Expected ASGI message 'http.response.body', but got '%s'." - raise RuntimeError(msg % message_type) - - body = cast(bytes, message.get("body", b"")) - more_body = message.get("more_body", False) - - # Write response body - if self.scope["method"] == "HEAD": - self.expected_content_length = 0 - elif self.chunked_encoding: - if body: - content = [b"%x\r\n" % len(body), body, b"\r\n"] - else: - content = [] - if not more_body: - content.append(b"0\r\n\r\n") - self.transport.write(b"".join(content)) - else: - num_bytes = len(body) - if num_bytes > self.expected_content_length: - raise RuntimeError("Response content longer than Content-Length") - else: - self.expected_content_length -= num_bytes - self.transport.write(body) - - # Handle response completion - if not more_body: - if self.expected_content_length != 0: - raise RuntimeError("Response content shorter than Content-Length") - self.response_complete = True - self.message_event.set() - if not self.keep_alive: - self.transport.close() - self.on_response() - - else: - # Response already sent - msg = "Unexpected ASGI message '%s' sent, after response already completed." - raise RuntimeError(msg % message_type) - - def connection_lost(self, exc: Exception | None) -> None: - self.connections.discard(self) - - if self.logger.level <= TRACE_LOG_LEVEL: - prefix = "%s:%d - " % self.client if self.client else "" - self.logger.log(TRACE_LOG_LEVEL, "%sHTTP connection lost", prefix) - - if self.cycle and not self.cycle.response_complete: - self.cycle.disconnected = True - if self.cycle is not None: - self.cycle.message_event.set() - if self.flow is not None: - self.flow.resume_writing() - if exc is None: - self.transport.close() - self._unset_keepalive_if_required() - - self.parser = None - diff --git a/routstr/auth.py b/routstr/auth.py index b7c8e16a..87eb60fd 100644 --- a/routstr/auth.py +++ b/routstr/auth.py @@ -695,8 +695,12 @@ async def pay_for_request( reserved_msats=reservation.reserved_msats, status="active", started_at=reserved_at_now, + # reserved_at_now floors to the second; add 1s margin so a + # finalizer finishing right at the nominal deadline isn't fenced + # out by truncation. expires_at=reserved_at_now - + math.ceil(remaining_lifetime + settings.request_cleanup_timeout_seconds), + + math.ceil(remaining_lifetime + settings.request_cleanup_timeout_seconds) + + 1, ) ) # Publish the identity before commit. If the commit succeeds but its @@ -737,11 +741,6 @@ async def pay_for_request( # The reservation is durable; keep its lease fresh for the whole request # lifetime (upstream header waits, non-streaming and streaming alike). - from .core.lifecycle import request_lifetime - - lifetime = request_lifetime.get() - if lifetime is not None: - lifetime.reservations.append(reservation) _start_reservation_heartbeat(reservation) try: diff --git a/routstr/core/lifecycle.py b/routstr/core/lifecycle.py index dad968ba..039ac046 100644 --- a/routstr/core/lifecycle.py +++ b/routstr/core/lifecycle.py @@ -4,11 +4,7 @@ from __future__ import annotations import asyncio from contextvars import ContextVar -from dataclasses import dataclass, field -from typing import TYPE_CHECKING - -if TYPE_CHECKING: - from ..auth import ReservationSnapshot +from dataclasses import dataclass from starlette.types import ASGIApp, Message, Receive, Scope, Send @@ -18,11 +14,14 @@ from .settings import settings logger = get_logger(__name__) +class DownstreamTerminated(OSError): + """Raised by downstream_send after disconnect; expected, not a server error.""" + + @dataclass class RequestLifetime: deadline: float = 0 stopped: bool = False - reservations: list[ReservationSnapshot] = field(default_factory=list) request_lifetime: ContextVar[RequestLifetime | None] = ContextVar( @@ -75,7 +74,7 @@ class RequestLifecycleMiddleware: async def downstream_send(message: Message) -> None: nonlocal response_started if disconnected.is_set() or lifetime.stopped: - raise OSError("Downstream request terminated") + raise DownstreamTerminated("Downstream request terminated") async with asyncio.timeout(settings.downstream_send_timeout_seconds): await send(message) if message["type"] == "http.response.start": @@ -86,27 +85,35 @@ class RequestLifecycleMiddleware: self.app(scope, downstream_receive, downstream_send) ) gone = asyncio.create_task(disconnected.wait()) + timed_out = False try: done, _ = await asyncio.wait( (work, gone), timeout=settings.max_request_lifetime_seconds, return_when=asyncio.FIRST_COMPLETED, ) + if gone in done and not response_started and work not in done: + # A pre-response wallet or billing operation may have accepted + # funds already. Let it reach its own settlement before closing. + done, _ = await asyncio.wait( + (work,), + timeout=max(0, lifetime.deadline - asyncio.get_running_loop().time()), + ) if work in done: - await work - elif not disconnected.is_set() and not response_started: - await downstream_send( - {"type": "http.response.start", "status": 504, "headers": []} - ) - await downstream_send( - {"type": "http.response.body", "body": b"Request deadline exceeded"} - ) + try: + await work + except DownstreamTerminated: + if not disconnected.is_set(): + raise + logger.debug("Client disconnected before response completed") + elif not disconnected.is_set(): + timed_out = True finally: lifetime.stopped = True for task in (receiver, gone, work): task.cancel() - # Cancellation/close is bounded: an uncooperative finalizer must not - # hold ownership or renewal indefinitely. + # Detached stream finalizers own settlement. The heartbeat stops + # with the request; durable expiry recovers any abandoned row. done, pending = await asyncio.wait( (receiver, gone, work), timeout=settings.request_cleanup_timeout_seconds ) @@ -118,20 +125,10 @@ class RequestLifecycleMiddleware: task.add_done_callback( lambda t: t.exception() if not t.cancelled() else None ) - try: - async with asyncio.timeout(settings.request_cleanup_timeout_seconds): - from ..auth import _stop_reservation_heartbeat, release_reservation - from .db import create_session - - for snapshot in lifetime.reservations: - await _stop_reservation_heartbeat(snapshot.release_id) - async with create_session() as session: - await release_reservation( - snapshot, session, snapshot.reserved_msats - ) - except Exception: - logger.exception( - "Request cleanup failed; durable expiry will recover reservations" + request_lifetime.reset(token) + if timed_out and work.done() and not disconnected.is_set() and not response_started: + async with asyncio.timeout(settings.downstream_send_timeout_seconds): + await send({"type": "http.response.start", "status": 504, "headers": []}) + await send( + {"type": "http.response.body", "body": b"Request deadline exceeded"} ) - finally: - request_lifetime.reset(token) diff --git a/tests/unit/test_request_lifecycle.py b/tests/unit/test_request_lifecycle.py index 41c76e50..1573f074 100644 --- a/tests/unit/test_request_lifecycle.py +++ b/tests/unit/test_request_lifecycle.py @@ -1,10 +1,30 @@ import asyncio +from collections.abc import AsyncGenerator +from contextlib import asynccontextmanager +from pathlib import Path from unittest.mock import patch import pytest +from sqlalchemy.ext.asyncio import create_async_engine +from sqlalchemy.pool import NullPool +from sqlmodel import SQLModel +from sqlmodel.ext.asyncio.session import AsyncSession +from starlette.applications import Starlette +from starlette.requests import Request +from starlette.responses import PlainTextResponse +from starlette.routing import Route from starlette.types import Message, Receive, Scope, Send +import routstr.core.db as db_module +from routstr.auth import ( + ReservationSnapshot, + _claim_reservation_for_charge, + _stop_reservation_heartbeat, + pay_for_request, +) +from routstr.core.db import ApiKey, ReservationRelease from routstr.core.lifecycle import RequestLifecycleMiddleware +from routstr.core.middleware import LoggingMiddleware from routstr.core.settings import settings @@ -56,3 +76,210 @@ async def test_lifecycle_stops_live_work(reason: str) -> None: await task assert closed.is_set() assert sent + + +@pytest.mark.asyncio +async def test_unrelated_oserror_still_propagates() -> None: + receive_queue: asyncio.Queue[Message] = asyncio.Queue() + await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) + + async def app(scope: Scope, receive: Receive, send: Send) -> None: + await receive() + raise OSError("Connection reset by peer") + + async def send(message: Message) -> None: + pass + + with pytest.raises(OSError, match="Connection reset by peer"): + await asyncio.wait_for( + RequestLifecycleMiddleware(app)({"type": "http"}, receive_queue.get, send), + 1, + ) + + +@pytest.mark.asyncio +async def test_disconnect_before_headers_preserves_wallet_work() -> None: + receive_queue: asyncio.Queue[Message] = asyncio.Queue() + await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) + entered = asyncio.Event() + finish_wallet = asyncio.Event() + wallet_credited = asyncio.Event() + + async def app(scope: Scope, receive: Receive, send: Send) -> None: + await receive() + entered.set() + await finish_wallet.wait() # The mint accepted the token; credit is still pending. + wallet_credited.set() + await send({"type": "http.response.start", "status": 200, "headers": []}) + + run = asyncio.create_task( + RequestLifecycleMiddleware(app)( + {"type": "http"}, receive_queue.get, lambda message: asyncio.sleep(0) + ) + ) + await asyncio.wait_for(entered.wait(), 1) + await receive_queue.put({"type": "http.disconnect"}) + await asyncio.sleep(0.02) + assert not run.done() + finish_wallet.set() + await asyncio.wait_for(run, 1) # No propagated exception for an expected disconnect. + assert wallet_credited.is_set() + + +@pytest.mark.asyncio +async def test_disconnect_before_headers_with_logging_middleware() -> None: + receive_queue: asyncio.Queue[Message] = asyncio.Queue() + await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) + entered = asyncio.Event() + finish_wallet = asyncio.Event() + wallet_credited = asyncio.Event() + + async def wallet(request: Request) -> PlainTextResponse: + await request.body() + entered.set() + await finish_wallet.wait() + wallet_credited.set() + return PlainTextResponse("settled") + + app = RequestLifecycleMiddleware( + LoggingMiddleware(Starlette(routes=[Route("/wallet", wallet, methods=["POST"])])) + ) + scope: Scope = { + "type": "http", + "asgi": {"version": "3.0", "spec_version": "2.4"}, + "http_version": "1.1", + "method": "POST", + "scheme": "http", + "path": "/wallet", + "raw_path": b"/wallet", + "root_path": "", + "query_string": b"", + "headers": [], + "client": ("test", 1234), + "server": ("test", 80), + } + + async def send(message: Message) -> None: + pass + + run = asyncio.create_task(app(scope, receive_queue.get, send)) + try: + await asyncio.wait_for(entered.wait(), 1) + await receive_queue.put({"type": "http.disconnect"}) + await asyncio.sleep(0.02) + assert not run.done() + finish_wallet.set() + await asyncio.wait_for(run, 1) # No propagated exception for an expected disconnect. + assert wallet_credited.is_set() + finally: + finish_wallet.set() + if not run.done(): + run.cancel() + await asyncio.gather(run, return_exceptions=True) + + +@pytest.mark.asyncio +async def test_deadline_cancels_app_before_sending_504() -> None: + receive_queue: asyncio.Queue[Message] = asyncio.Queue() + await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) + sent: list[Message] = [] + app_stopped = asyncio.Event() + + async def app(scope: Scope, receive: Receive, send: Send) -> None: + await receive() + try: + await asyncio.sleep(100) + finally: + with pytest.raises(OSError, match="Downstream request terminated"): + await send({"type": "http.response.start", "status": 200, "headers": []}) + app_stopped.set() + + async def send(message: Message) -> None: + assert app_stopped.is_set() + sent.append(message) + + with ( + patch.object(settings, "max_request_lifetime_seconds", 0.02), + patch.object(settings, "request_cleanup_timeout_seconds", 0.1), + ): + await asyncio.wait_for( + RequestLifecycleMiddleware(app)({"type": "http"}, receive_queue.get, send), + 1, + ) + assert [message["type"] for message in sent] == [ + "http.response.start", + "http.response.body", + ] + assert sent[0]["status"] == 504 + + +@pytest.mark.asyncio +async def test_disconnect_does_not_release_before_stream_settles(tmp_path: Path) -> None: + engine = create_async_engine( + f"sqlite+aiosqlite:///{tmp_path / 'reservations.db'}", poolclass=NullPool + ) + async with engine.begin() as conn: + await conn.run_sync(SQLModel.metadata.create_all) + + @asynccontextmanager + async def session() -> AsyncGenerator[AsyncSession, None]: + async with AsyncSession(engine, expire_on_commit=False) as db: + yield db + + with patch.object(db_module, "create_session", session): + async with session() as db: + db.add(ApiKey(hashed_key="stream-key", balance=10_000)) + await db.commit() + started = asyncio.Event() + finalizer_started = asyncio.Event() + settle = asyncio.Event() + result: asyncio.Future[bool] = asyncio.get_running_loop().create_future() + snapshot: ReservationSnapshot | None = None + receive_queue: asyncio.Queue[Message] = asyncio.Queue() + await receive_queue.put({"type": "http.request", "body": b"", "more_body": False}) + + async def app(scope: Scope, receive: Receive, send: Send) -> None: + nonlocal snapshot + async with session() as db: + key = await db.get(ApiKey, "stream-key") + assert key is not None + snapshot = await pay_for_request(key, 1000, db) + await receive() + await send({"type": "http.response.start", "status": 200, "headers": []}) + started.set() + try: + await asyncio.sleep(100) + finally: + async def finalize() -> None: + assert snapshot is not None + finalizer_started.set() + await settle.wait() + async with session() as db: + claimed = await _claim_reservation_for_charge(snapshot, db) + await db.commit() + await _stop_reservation_heartbeat(snapshot.release_id) + result.set_result(claimed) + + asyncio.create_task(finalize()) + + run = asyncio.create_task( + RequestLifecycleMiddleware(app)( + {"type": "http"}, receive_queue.get, lambda message: asyncio.sleep(0) + ) + ) + try: + await asyncio.wait_for(started.wait(), 1) + await receive_queue.put({"type": "http.disconnect"}) + await asyncio.wait_for(finalizer_started.wait(), 1) + await asyncio.wait_for(run, 1) + settle.set() + assert await asyncio.wait_for(result, 1) + assert snapshot is not None + async with session() as db: + row = await db.get(ReservationRelease, snapshot.release_id) + assert row is not None and row.status == "charged" + finally: + settle.set() + if snapshot is not None: + await _stop_reservation_heartbeat(snapshot.release_id) + await engine.dispose() diff --git a/tests/unit/test_stale_reservations.py b/tests/unit/test_stale_reservations.py index 6cd59682..61a3da5a 100644 --- a/tests/unit/test_stale_reservations.py +++ b/tests/unit/test_stale_reservations.py @@ -9,6 +9,7 @@ Covers: """ import asyncio +import math import time from typing import AsyncGenerator from unittest.mock import AsyncMock, MagicMock, patch @@ -69,7 +70,6 @@ async def test_pay_for_request_sets_reserved_at( payments_info = MagicMock() monkeypatch.setattr(auth_module.logger, "info", logger_info) monkeypatch.setattr(auth_module.payments_logger, "info", payments_info) - before = int(time.time()) await pay_for_request(key, 1_000, session) @@ -87,6 +87,34 @@ async def test_pay_for_request_sets_reserved_at( assert payments_info.call_args.args == ("RESERVE",) +@pytest.mark.asyncio +async def test_pay_for_request_expires_at_has_floor_margin( + session: AsyncSession, monkeypatch: pytest.MonkeyPatch +) -> None: + """reserved_at_now floors to the second; expires_at must add 1s so a + finalizer finishing exactly at the nominal deadline isn't fenced out.""" + key = ApiKey(hashed_key="floorkey", balance=10_000) + session.add(key) + await session.commit() + + fixed_time = 1_700_000_000.9 # fractional second, floors when int()'d + monkeypatch.setattr(auth_module.time, "time", lambda: fixed_time) + + snapshot = await pay_for_request(key, 1_000, session) + + row = await session.get(ReservationRelease, snapshot.release_id) + assert row is not None + expected = ( + int(fixed_time) + + math.ceil( + auth_module.settings.max_request_lifetime_seconds + + auth_module.settings.request_cleanup_timeout_seconds + ) + + 1 + ) + assert row.expires_at == expected + + @pytest.mark.asyncio @pytest.mark.asyncio async def test_pay_for_request_releases_reservation_when_validation_fails(