diff --git a/src/daemon/http/index.ts b/src/daemon/http/index.ts index cbf9d2e..e49bedc 100644 --- a/src/daemon/http/index.ts +++ b/src/daemon/http/index.ts @@ -325,6 +325,8 @@ export function createDaemonRequestHandler(deps: { getModelProviders: (modelId: string) => Promise; refreshProvidersAndModels: () => Promise; mode?: "xcashu" | "apikeys"; + /** Default max_tokens/max_output_tokens to inject when the client omits one. */ + maxTokens: number; /** Nostr hex pubkey for routstr review/model events (kind 38425/38423). */ routstrPubkey?: string; usageTrackingDriver: UsageTrackingDriver; @@ -1486,6 +1488,22 @@ export function createDaemonRequestHandler(deps: { return; } + // Cap the completion budget when the client does not set an output token + // limit. Without this, the SDK prices at the provider's worst-case + // max_completion_cost, which varies widely across providers (2.3× for + // kimi-k3) and balloons during provider failover. Chat/completions use + // max_tokens; the OpenAI Responses API uses max_output_tokens. + if (deps.maxTokens > 0) { + const isResponsesPath = url.pathname.includes("/responses"); + if (isResponsesPath) { + if (typeof bodyObj.max_output_tokens !== "number") { + bodyObj.max_output_tokens = deps.maxTokens; + } + } else if (typeof bodyObj.max_tokens !== "number") { + bodyObj.max_tokens = deps.maxTokens; + } + } + const forcedProvider: string | undefined = url.searchParams.get("provider") || (req.headers["x-routstr-provider"] as string | undefined) || diff --git a/src/daemon/index.ts b/src/daemon/index.ts index 6475874..ed5021d 100644 --- a/src/daemon/index.ts +++ b/src/daemon/index.ts @@ -217,6 +217,7 @@ async function main(): Promise { getModelProviders, refreshProvidersAndModels, mode: config.mode || "apikeys", + maxTokens: config.maxTokens ?? 64000, routstrPubkey: config.routstrPubkey, usageTrackingDriver, providerManager, diff --git a/src/utils/config.ts b/src/utils/config.ts index 2ebe28f..49a193e 100644 --- a/src/utils/config.ts +++ b/src/utils/config.ts @@ -55,6 +55,14 @@ export interface RoutstrdConfig { relays?: string[]; /** NWC integration configuration */ nwc?: NwcConfig; + /** + * Default max_tokens (chat/completions) and max_output_tokens (responses) + * injected into proxied requests when the client does not supply one. + * Caps the completion budget the SDK prices against (completion × maxTokens) + * instead of the provider's worst-case max_completion_cost. Set to 0 to + * disable injection and always forward the client's value as-is. + */ + maxTokens?: number; } export const DEFAULT_CONFIG: RoutstrdConfig = { @@ -63,4 +71,5 @@ export const DEFAULT_CONFIG: RoutstrdConfig = { provider: null, cocodPath: null, mode: "apikeys", + maxTokens: 64000, };