mirror of
https://github.com/Routstr/routstrd.git
synced 2026-10-05 12:28:23 +00:00
fix(proxy): inject max_completion_tokens instead of max_tokens
When the client omits an output-token cap, routstrd injects one so the SDK prices against a bounded completion budget instead of the provider's worst-case max_completion_cost. It previously injected the legacy max_tokens field, which is deprecated in the OpenAI chat-completions spec and rejected by reasoning models. Inject max_completion_tokens (the OpenAI-standard field) instead, and only when neither max_completion_tokens nor max_tokens is already present, so an explicit client cap is never overridden. The Responses API path is unchanged (max_output_tokens).
This commit is contained in:
@@ -1712,15 +1712,20 @@ export function createDaemonRequestHandler(deps: {
|
||||
// limit. Without this, the SDK prices at the provider's worst-case
|
||||
// max_completion_cost, which varies widely across providers (2.3× for
|
||||
// kimi-k3) and balloons during provider failover. Chat/completions use
|
||||
// max_tokens; the OpenAI Responses API uses max_output_tokens.
|
||||
// max_completion_tokens (the OpenAI-standard field; legacy max_tokens is
|
||||
// deprecated and rejected by reasoning models); the Responses API uses
|
||||
// max_output_tokens.
|
||||
if (deps.maxTokens > 0) {
|
||||
const isResponsesPath = url.pathname.includes("/responses");
|
||||
if (isResponsesPath) {
|
||||
if (typeof bodyObj.max_output_tokens !== "number") {
|
||||
bodyObj.max_output_tokens = deps.maxTokens;
|
||||
}
|
||||
} else if (typeof bodyObj.max_tokens !== "number") {
|
||||
bodyObj.max_tokens = deps.maxTokens;
|
||||
} else if (
|
||||
typeof bodyObj.max_completion_tokens !== "number" &&
|
||||
typeof bodyObj.max_tokens !== "number"
|
||||
) {
|
||||
bodyObj.max_completion_tokens = deps.maxTokens;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user