mirror of
https://github.com/Routstr/routstrd.git
synced 2026-10-05 12:28:23 +00:00
revert: use max_tokens for chat/completions instead of max_completion_tokens
Revert the default request-limit field back to legacy max_tokens for chat/completions (keeping max_output_tokens for the Responses API).
This commit is contained in:
@@ -1712,20 +1712,15 @@ export function createDaemonRequestHandler(deps: {
|
|||||||
// limit. Without this, the SDK prices at the provider's worst-case
|
// limit. Without this, the SDK prices at the provider's worst-case
|
||||||
// max_completion_cost, which varies widely across providers (2.3× for
|
// max_completion_cost, which varies widely across providers (2.3× for
|
||||||
// kimi-k3) and balloons during provider failover. Chat/completions use
|
// kimi-k3) and balloons during provider failover. Chat/completions use
|
||||||
// max_completion_tokens (the OpenAI-standard field; legacy max_tokens is
|
// max_tokens; the OpenAI Responses API uses max_output_tokens.
|
||||||
// deprecated and rejected by reasoning models); the Responses API uses
|
|
||||||
// max_output_tokens.
|
|
||||||
if (deps.maxTokens > 0) {
|
if (deps.maxTokens > 0) {
|
||||||
const isResponsesPath = url.pathname.includes("/responses");
|
const isResponsesPath = url.pathname.includes("/responses");
|
||||||
if (isResponsesPath) {
|
if (isResponsesPath) {
|
||||||
if (typeof bodyObj.max_output_tokens !== "number") {
|
if (typeof bodyObj.max_output_tokens !== "number") {
|
||||||
bodyObj.max_output_tokens = deps.maxTokens;
|
bodyObj.max_output_tokens = deps.maxTokens;
|
||||||
}
|
}
|
||||||
} else if (
|
} else if (typeof bodyObj.max_tokens !== "number") {
|
||||||
typeof bodyObj.max_completion_tokens !== "number" &&
|
bodyObj.max_tokens = deps.maxTokens;
|
||||||
typeof bodyObj.max_tokens !== "number"
|
|
||||||
) {
|
|
||||||
bodyObj.max_completion_tokens = deps.maxTokens;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user