From 0aec14e001ebff0560403203e8d5ba7f24d70c0e Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:32:54 +0300 Subject: [PATCH] fix(http): inject max_tokens only on OpenAI-compatible endpoints MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The ingress added `max_tokens` (or `max_output_tokens` on /responses) to every proxied POST body that omitted one, defaulting to the configured 64000. That field is OpenAI chat vocabulary, and strict endpoints reject it: TypeSafe's `POST /v1/systemone` validates its body and answers `400 Invalid request.` for any unknown top-level key, so a valid System One request failed whenever `maxTokens > 0` — the default. Gate the injection on the endpoints that define the field, mirroring `isOpenAiJsonBodyPath` in @routstr/sdk. `/v1/systemone`, `/v1/embeddings`, `/v1/audio/*` and `/v1/messages` are now forwarded verbatim. Pricing is unchanged where the model envelope has no completion price: `getRequiredSatsForModel` falls back to `max_cost`, which for jev-latest (completion 0, max_cost 3.379) is the same reserve the capped budget produced. The SDK injects `stream: false` into every object body and needs the matching fix (Routstr/routstr-sdk#63); with either injection present the upstream request is still a 400. --- src/daemon/http/index.ts | 16 +++---- src/daemon/http/request-body.test.ts | 62 +++++++++++++++++++++++++ src/daemon/http/request-body.ts | 67 ++++++++++++++++++++++++++++ 3 files changed, 134 insertions(+), 11 deletions(-) create mode 100644 src/daemon/http/request-body.test.ts create mode 100644 src/daemon/http/request-body.ts diff --git a/src/daemon/http/index.ts b/src/daemon/http/index.ts index 510d239..d055e63 100644 --- a/src/daemon/http/index.ts +++ b/src/daemon/http/index.ts @@ -21,6 +21,7 @@ import { import { receiveCashuToken } from "../wallet"; import { getClientsFromStore } from "../../utils/clients"; import { getUsageSummary } from "./usage-summary"; +import { applyDefaultOutputTokenLimit } from "./request-body"; // Hop-by-hop headers describe the *upstream* connection, not this one, and must // never be copied onto our response. In particular, copying the upstream's @@ -1747,17 +1748,10 @@ export function createDaemonRequestHandler(deps: { // limit. Without this, the SDK prices at the provider's worst-case // max_completion_cost, which varies widely across providers (2.3× for // kimi-k3) and balloons during provider failover. Chat/completions use - // max_tokens; the OpenAI Responses API uses max_output_tokens. - if (deps.maxTokens > 0) { - const isResponsesPath = url.pathname.includes("/responses"); - if (isResponsesPath) { - if (typeof bodyObj.max_output_tokens !== "number") { - bodyObj.max_output_tokens = deps.maxTokens; - } - } else if (typeof bodyObj.max_tokens !== "number") { - bodyObj.max_tokens = deps.maxTokens; - } - } + // max_tokens; the OpenAI Responses API uses max_output_tokens. Endpoints + // that do not define those fields (e.g. /v1/systemone) are forwarded + // verbatim — injecting one makes a strict upstream answer 400. + applyDefaultOutputTokenLimit(url.pathname, bodyObj, deps.maxTokens); const forcedProvider: string | undefined = url.searchParams.get("provider") || diff --git a/src/daemon/http/request-body.test.ts b/src/daemon/http/request-body.test.ts new file mode 100644 index 0000000..ff0dfcd --- /dev/null +++ b/src/daemon/http/request-body.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from "bun:test"; +import { + OPENAI_JSON_BODY_PATH_SUFFIXES, + applyDefaultOutputTokenLimit, + isOpenAiJsonBodyPath, +} from "./request-body"; + +const MAX_TOKENS = 64000; + +describe("isOpenAiJsonBodyPath", () => { + it("accepts OpenAI-compatible endpoints, with or without a prefix/query/slash", () => { + for (const suffix of OPENAI_JSON_BODY_PATH_SUFFIXES) { + expect(isOpenAiJsonBodyPath(`/v1${suffix}`)).toBe(true); + expect(isOpenAiJsonBodyPath(`/proxy/v1${suffix}?trace=1`)).toBe(true); + expect(isOpenAiJsonBodyPath(`/v1${suffix}/`)).toBe(true); + } + }); + + it("rejects endpoints that do not define the OpenAI request vocabulary", () => { + for (const path of ["/v1/systemone", "/v1/embeddings", "/v1/audio/speech", "/v1/messages"]) { + expect(isOpenAiJsonBodyPath(path)).toBe(false); + } + }); +}); + +describe("applyDefaultOutputTokenLimit", () => { + it("injects max_tokens on chat/completions", () => { + const body: Record = { model: "glm-5.3-flash", messages: [] }; + expect(applyDefaultOutputTokenLimit("/v1/chat/completions", body, MAX_TOKENS)).toBe(true); + expect(body.max_tokens).toBe(MAX_TOKENS); + }); + + it("injects max_output_tokens on /responses", () => { + const body: Record = { model: "gpt-5.4", input: "hi" }; + expect(applyDefaultOutputTokenLimit("/v1/responses", body, MAX_TOKENS)).toBe(true); + expect(body.max_output_tokens).toBe(MAX_TOKENS); + expect(body).not.toHaveProperty("max_tokens"); + }); + + it("keeps a client-supplied limit", () => { + const body: Record = { messages: [], max_tokens: 5 }; + expect(applyDefaultOutputTokenLimit("/v1/chat/completions", body, MAX_TOKENS)).toBe(false); + expect(body.max_tokens).toBe(5); + }); + + it("leaves /v1/systemone untouched — the upstream rejects unknown fields", () => { + const body: Record = { + state: "Help! My payouts have been failing for 3 days.", + model: "jev-latest", + questions: { is_urgent: { type: "noul", instructions: "Does this convey urgency?" } }, + }; + const before = structuredClone(body); + expect(applyDefaultOutputTokenLimit("/v1/systemone", body, MAX_TOKENS)).toBe(false); + expect(body).toEqual(before); + }); + + it("is a no-op when injection is disabled", () => { + const body: Record = { messages: [] }; + expect(applyDefaultOutputTokenLimit("/v1/chat/completions", body, 0)).toBe(false); + expect(body).not.toHaveProperty("max_tokens"); + }); +}); diff --git a/src/daemon/http/request-body.ts b/src/daemon/http/request-body.ts new file mode 100644 index 0000000..9955a44 --- /dev/null +++ b/src/daemon/http/request-body.ts @@ -0,0 +1,67 @@ +/** + * Request-body shaping for the proxied path. + * + * The ingress forwards the caller's JSON body to the upstream provider. It is + * allowed to add exactly one thing — a default output-token limit — and only + * where that field is part of the endpoint's documented vocabulary. Anything + * else must be forwarded verbatim: several upstreams (notably TypeSafe's + * `POST /v1/systemone`) validate strictly and reject unknown fields, so an + * injected chat-completions field turns a valid request into an opaque + * `400 Invalid request.` + */ + +/** + * OpenAI-compatible endpoints whose bodies define `max_tokens`, + * `max_output_tokens`, and `stream`. + * + * Matched by path suffix so `/v1/chat/completions`, `/chat/completions`, and + * custom path-prefixed proxies all qualify. Kept in sync with + * `isOpenAiJsonBodyPath` in `@routstr/sdk` (exported there since 0.4.6); once + * routstrd depends on that version this local copy can be dropped. + */ +export const OPENAI_JSON_BODY_PATH_SUFFIXES = [ + "/chat/completions", + "/completions", + "/responses", +] as const; + +/** + * True when `pathname` addresses an endpoint whose body carries the OpenAI + * request vocabulary. Ignores a query string and a trailing slash. + */ +export function isOpenAiJsonBodyPath(pathname: string): boolean { + const path = (pathname.split("?")[0] ?? "").replace(/\/+$/, ""); + return OPENAI_JSON_BODY_PATH_SUFFIXES.some((suffix) => path.endsWith(suffix)); +} + +/** + * Cap the completion budget when the client does not set an output token + * limit. Without this, the SDK prices at the provider's worst-case + * `max_completion_cost`, which varies widely across providers (2.3× for + * kimi-k3) and balloons during provider failover. Chat/completions use + * `max_tokens`; the OpenAI Responses API uses `max_output_tokens`. + * + * Non-OpenAI endpoints are left untouched — `/v1/systemone` returns typed + * judgments rather than generated text, so it has no completion budget, and it + * rejects the field outright. Mutates `body` in place. Returns true when a + * field was added. + */ +export function applyDefaultOutputTokenLimit( + pathname: string, + body: Record, + maxTokens: number, +): boolean { + if (!(maxTokens > 0) || !isOpenAiJsonBodyPath(pathname)) { + return false; + } + + if (pathname.split("?")[0]!.replace(/\/+$/, "").endsWith("/responses")) { + if (typeof body.max_output_tokens === "number") return false; + body.max_output_tokens = maxTokens; + return true; + } + + if (typeof body.max_tokens === "number") return false; + body.max_tokens = maxTokens; + return true; +}