revert: use max_tokens for chat/completions instead of max_completion_tokens

Revert the default request-limit field back to legacy max_tokens for
chat/completions (keeping max_output_tokens for the Responses API).
This commit is contained in:
redshift
2026-09-07 14:11:52 +02:00
parent 6499bbd870
commit 6218b20bf9
+3 -8
View File
@@ -1712,20 +1712,15 @@ export function createDaemonRequestHandler(deps: {
// limit. Without this, the SDK prices at the provider's worst-case // limit. Without this, the SDK prices at the provider's worst-case
// max_completion_cost, which varies widely across providers (2.3× for // max_completion_cost, which varies widely across providers (2.3× for
// kimi-k3) and balloons during provider failover. Chat/completions use // kimi-k3) and balloons during provider failover. Chat/completions use
// max_completion_tokens (the OpenAI-standard field; legacy max_tokens is // max_tokens; the OpenAI Responses API uses max_output_tokens.
// deprecated and rejected by reasoning models); the Responses API uses
// max_output_tokens.
if (deps.maxTokens > 0) { if (deps.maxTokens > 0) {
const isResponsesPath = url.pathname.includes("/responses"); const isResponsesPath = url.pathname.includes("/responses");
if (isResponsesPath) { if (isResponsesPath) {
if (typeof bodyObj.max_output_tokens !== "number") { if (typeof bodyObj.max_output_tokens !== "number") {
bodyObj.max_output_tokens = deps.maxTokens; bodyObj.max_output_tokens = deps.maxTokens;
} }
} else if ( } else if (typeof bodyObj.max_tokens !== "number") {
typeof bodyObj.max_completion_tokens !== "number" && bodyObj.max_tokens = deps.maxTokens;
typeof bodyObj.max_tokens !== "number"
) {
bodyObj.max_completion_tokens = deps.maxTokens;
} }
} }