From 6218b20bf9474fb209dc37279f4d3a92c2f137fe Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Mon, 7 Sep 2026 14:11:52 +0200 Subject: [PATCH] revert: use max_tokens for chat/completions instead of max_completion_tokens Revert the default request-limit field back to legacy max_tokens for chat/completions (keeping max_output_tokens for the Responses API). --- src/daemon/http/index.ts | 11 +++-------- 1 file changed, 3 insertions(+), 8 deletions(-) diff --git a/src/daemon/http/index.ts b/src/daemon/http/index.ts index 354d42f..505b6c8 100644 --- a/src/daemon/http/index.ts +++ b/src/daemon/http/index.ts @@ -1712,20 +1712,15 @@ export function createDaemonRequestHandler(deps: { // limit. Without this, the SDK prices at the provider's worst-case // max_completion_cost, which varies widely across providers (2.3× for // kimi-k3) and balloons during provider failover. Chat/completions use - // max_completion_tokens (the OpenAI-standard field; legacy max_tokens is - // deprecated and rejected by reasoning models); the Responses API uses - // max_output_tokens. + // max_tokens; the OpenAI Responses API uses max_output_tokens. if (deps.maxTokens > 0) { const isResponsesPath = url.pathname.includes("/responses"); if (isResponsesPath) { if (typeof bodyObj.max_output_tokens !== "number") { bodyObj.max_output_tokens = deps.maxTokens; } - } else if ( - typeof bodyObj.max_completion_tokens !== "number" && - typeof bodyObj.max_tokens !== "number" - ) { - bodyObj.max_completion_tokens = deps.maxTokens; + } else if (typeof bodyObj.max_tokens !== "number") { + bodyObj.max_tokens = deps.maxTokens; } }