Merge pull request #108 from Routstr/fix/max-tokens-injection-openai-only

fix(http): inject max_tokens only on OpenAI-compatible endpoints
This commit is contained in:
redshift
2026-09-20 17:47:39 +02:00
committed by GitHub
3 changed files with 134 additions and 11 deletions
+5 -11
View File
@@ -21,6 +21,7 @@ import {
import { receiveCashuToken } from "../wallet";
import { getClientsFromStore } from "../../utils/clients";
import { getUsageSummary } from "./usage-summary";
import { applyDefaultOutputTokenLimit } from "./request-body";
// Hop-by-hop headers describe the *upstream* connection, not this one, and must
// never be copied onto our response. In particular, copying the upstream's
@@ -1747,17 +1748,10 @@ export function createDaemonRequestHandler(deps: {
// limit. Without this, the SDK prices at the provider's worst-case
// max_completion_cost, which varies widely across providers (2.3× for
// kimi-k3) and balloons during provider failover. Chat/completions use
// max_tokens; the OpenAI Responses API uses max_output_tokens.
if (deps.maxTokens > 0) {
const isResponsesPath = url.pathname.includes("/responses");
if (isResponsesPath) {
if (typeof bodyObj.max_output_tokens !== "number") {
bodyObj.max_output_tokens = deps.maxTokens;
}
} else if (typeof bodyObj.max_tokens !== "number") {
bodyObj.max_tokens = deps.maxTokens;
}
}
// max_tokens; the OpenAI Responses API uses max_output_tokens. Endpoints
// that do not define those fields (e.g. /v1/systemone) are forwarded
// verbatim — injecting one makes a strict upstream answer 400.
applyDefaultOutputTokenLimit(url.pathname, bodyObj, deps.maxTokens);
const forcedProvider: string | undefined =
url.searchParams.get("provider") ||
+62
View File
@@ -0,0 +1,62 @@
import { describe, expect, it } from "bun:test";
import {
OPENAI_JSON_BODY_PATH_SUFFIXES,
applyDefaultOutputTokenLimit,
isOpenAiJsonBodyPath,
} from "./request-body";
const MAX_TOKENS = 64000;
describe("isOpenAiJsonBodyPath", () => {
it("accepts OpenAI-compatible endpoints, with or without a prefix/query/slash", () => {
for (const suffix of OPENAI_JSON_BODY_PATH_SUFFIXES) {
expect(isOpenAiJsonBodyPath(`/v1${suffix}`)).toBe(true);
expect(isOpenAiJsonBodyPath(`/proxy/v1${suffix}?trace=1`)).toBe(true);
expect(isOpenAiJsonBodyPath(`/v1${suffix}/`)).toBe(true);
}
});
it("rejects endpoints that do not define the OpenAI request vocabulary", () => {
for (const path of ["/v1/systemone", "/v1/embeddings", "/v1/audio/speech", "/v1/messages"]) {
expect(isOpenAiJsonBodyPath(path)).toBe(false);
}
});
});
describe("applyDefaultOutputTokenLimit", () => {
it("injects max_tokens on chat/completions", () => {
const body: Record<string, unknown> = { model: "glm-5.3-flash", messages: [] };
expect(applyDefaultOutputTokenLimit("/v1/chat/completions", body, MAX_TOKENS)).toBe(true);
expect(body.max_tokens).toBe(MAX_TOKENS);
});
it("injects max_output_tokens on /responses", () => {
const body: Record<string, unknown> = { model: "gpt-5.4", input: "hi" };
expect(applyDefaultOutputTokenLimit("/v1/responses", body, MAX_TOKENS)).toBe(true);
expect(body.max_output_tokens).toBe(MAX_TOKENS);
expect(body).not.toHaveProperty("max_tokens");
});
it("keeps a client-supplied limit", () => {
const body: Record<string, unknown> = { messages: [], max_tokens: 5 };
expect(applyDefaultOutputTokenLimit("/v1/chat/completions", body, MAX_TOKENS)).toBe(false);
expect(body.max_tokens).toBe(5);
});
it("leaves /v1/systemone untouched — the upstream rejects unknown fields", () => {
const body: Record<string, unknown> = {
state: "Help! My payouts have been failing for 3 days.",
model: "jev-latest",
questions: { is_urgent: { type: "noul", instructions: "Does this convey urgency?" } },
};
const before = structuredClone(body);
expect(applyDefaultOutputTokenLimit("/v1/systemone", body, MAX_TOKENS)).toBe(false);
expect(body).toEqual(before);
});
it("is a no-op when injection is disabled", () => {
const body: Record<string, unknown> = { messages: [] };
expect(applyDefaultOutputTokenLimit("/v1/chat/completions", body, 0)).toBe(false);
expect(body).not.toHaveProperty("max_tokens");
});
});
+67
View File
@@ -0,0 +1,67 @@
/**
* Request-body shaping for the proxied path.
*
* The ingress forwards the caller's JSON body to the upstream provider. It is
* allowed to add exactly one thing — a default output-token limit — and only
* where that field is part of the endpoint's documented vocabulary. Anything
* else must be forwarded verbatim: several upstreams (notably TypeSafe's
* `POST /v1/systemone`) validate strictly and reject unknown fields, so an
* injected chat-completions field turns a valid request into an opaque
* `400 Invalid request.`
*/
/**
* OpenAI-compatible endpoints whose bodies define `max_tokens`,
* `max_output_tokens`, and `stream`.
*
* Matched by path suffix so `/v1/chat/completions`, `/chat/completions`, and
* custom path-prefixed proxies all qualify. Kept in sync with
* `isOpenAiJsonBodyPath` in `@routstr/sdk` (exported there since 0.4.6); once
* routstrd depends on that version this local copy can be dropped.
*/
export const OPENAI_JSON_BODY_PATH_SUFFIXES = [
"/chat/completions",
"/completions",
"/responses",
] as const;
/**
* True when `pathname` addresses an endpoint whose body carries the OpenAI
* request vocabulary. Ignores a query string and a trailing slash.
*/
export function isOpenAiJsonBodyPath(pathname: string): boolean {
const path = (pathname.split("?")[0] ?? "").replace(/\/+$/, "");
return OPENAI_JSON_BODY_PATH_SUFFIXES.some((suffix) => path.endsWith(suffix));
}
/**
* Cap the completion budget when the client does not set an output token
* limit. Without this, the SDK prices at the provider's worst-case
* `max_completion_cost`, which varies widely across providers (2.3× for
* kimi-k3) and balloons during provider failover. Chat/completions use
* `max_tokens`; the OpenAI Responses API uses `max_output_tokens`.
*
* Non-OpenAI endpoints are left untouched — `/v1/systemone` returns typed
* judgments rather than generated text, so it has no completion budget, and it
* rejects the field outright. Mutates `body` in place. Returns true when a
* field was added.
*/
export function applyDefaultOutputTokenLimit(
pathname: string,
body: Record<string, unknown>,
maxTokens: number,
): boolean {
if (!(maxTokens > 0) || !isOpenAiJsonBodyPath(pathname)) {
return false;
}
if (pathname.split("?")[0]!.replace(/\/+$/, "").endsWith("/responses")) {
if (typeof body.max_output_tokens === "number") return false;
body.max_output_tokens = maxTokens;
return true;
}
if (typeof body.max_tokens === "number") return false;
body.max_tokens = maxTokens;
return true;
}