mirror of
https://github.com/Kilo-Org/kilocode.git
synced 2026-09-24 16:02:55 +08:00
fix(cli): drop max_tokens for gpt-5 models on @ai-sdk/openai-compatible
gpt-5 models rejected via openai-compatible proxies (e.g. LiteLLM) because the SDK emits max_tokens while OpenAI requires max_completion_tokens and the compatible SDK cannot rename the field. Clear the cap so the upstream default output budget applies.
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
---
|
||||
"@kilocode/cli": patch
|
||||
---
|
||||
|
||||
Fix gpt-5 models routed through OpenAI-compatible providers (e.g. LiteLLM) rejecting requests with "Unsupported parameter: max_tokens". The CLI now drops the max output token cap for gpt-5 models on `@ai-sdk/openai-compatible`, letting the upstream default output budget apply.
|
||||
@@ -0,0 +1,17 @@
|
||||
import type { Hooks, PluginInput } from "@kilocode/plugin"
|
||||
|
||||
// Drop the max output tokens cap for gpt-5 models routed through
|
||||
// @ai-sdk/openai-compatible. The SDK always emits `max_tokens`, but OpenAI's
|
||||
// gpt-5 family (and compatible proxies like LiteLLM) rejects that field and
|
||||
// requires `max_completion_tokens` instead. The openai-compatible SDK has no
|
||||
// way to rename the field, so we clear the cap and let the upstream default
|
||||
// output budget apply.
|
||||
export async function DropMaxTokensForGpt5Plugin(_input: PluginInput): Promise<Hooks> {
|
||||
return {
|
||||
"chat.params": async (input, output) => {
|
||||
if (input.model.api.npm !== "@ai-sdk/openai-compatible") return
|
||||
if (!input.model.api.id.toLowerCase().includes("gpt-5")) return
|
||||
output.maxOutputTokens = undefined
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -25,6 +25,7 @@ import { errorMessage } from "@/util/error"
|
||||
import { PluginLoader } from "./loader"
|
||||
import { parsePluginSpecifier, readPluginId, readV1Plugin, resolvePluginId } from "./shared"
|
||||
import { KiloAuthPlugin } from "@kilocode/kilo-gateway" // kilocode_change
|
||||
import { DropMaxTokensForGpt5Plugin } from "@/kilocode/plugin/drop-max-tokens" // kilocode_change
|
||||
import { registerAdapter } from "@/control-plane/adapters"
|
||||
import type { WorkspaceAdapter } from "@/control-plane/types"
|
||||
|
||||
@@ -67,6 +68,7 @@ const INTERNAL_PLUGINS: PluginInstance[] = [
|
||||
CloudflareWorkersAuthPlugin,
|
||||
CloudflareAIGatewayAuthPlugin,
|
||||
AzureAuthPlugin,
|
||||
DropMaxTokensForGpt5Plugin,
|
||||
]
|
||||
// kilocode_change end
|
||||
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
import { expect, test } from "bun:test"
|
||||
import { DropMaxTokensForGpt5Plugin } from "@/kilocode/plugin/drop-max-tokens"
|
||||
|
||||
const pluginInput = {
|
||||
client: {} as never,
|
||||
project: {} as never,
|
||||
directory: "",
|
||||
worktree: "",
|
||||
experimental_workspace: { register() {} },
|
||||
serverUrl: new URL("https://example.com"),
|
||||
$: {} as never,
|
||||
}
|
||||
|
||||
function makeHookInput(overrides: { providerID?: string; apiId?: string; npm?: string; reasoning?: boolean }) {
|
||||
return {
|
||||
sessionID: "s",
|
||||
agent: "a",
|
||||
provider: {} as never,
|
||||
message: {} as never,
|
||||
model: {
|
||||
providerID: overrides.providerID ?? "litellm",
|
||||
api: {
|
||||
id: overrides.apiId ?? "gpt-5",
|
||||
url: "",
|
||||
npm: overrides.npm ?? "@ai-sdk/openai-compatible",
|
||||
},
|
||||
capabilities: {
|
||||
reasoning: overrides.reasoning ?? true,
|
||||
temperature: false,
|
||||
attachment: true,
|
||||
toolcall: true,
|
||||
input: { text: true, audio: false, image: false, video: false, pdf: false },
|
||||
output: { text: true, audio: false, image: false, video: false, pdf: false },
|
||||
interleaved: false,
|
||||
},
|
||||
} as never,
|
||||
}
|
||||
}
|
||||
|
||||
function makeHookOutput() {
|
||||
return { temperature: 0, topP: 1, topK: 0, maxOutputTokens: 32_000 as number | undefined, options: {} }
|
||||
}
|
||||
|
||||
test("drops maxOutputTokens for gpt-5 on @ai-sdk/openai-compatible", async () => {
|
||||
const hooks = await DropMaxTokensForGpt5Plugin(pluginInput)
|
||||
const out = makeHookOutput()
|
||||
await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-5" }), out)
|
||||
expect(out.maxOutputTokens).toBeUndefined()
|
||||
})
|
||||
|
||||
test("drops maxOutputTokens for gpt-5 variants (codex, pro, mini) on openai-compatible", async () => {
|
||||
const hooks = await DropMaxTokensForGpt5Plugin(pluginInput)
|
||||
for (const id of ["gpt-5.2-codex", "gpt-5-pro", "gpt-5-mini", "openai/gpt-5-turbo"]) {
|
||||
const out = makeHookOutput()
|
||||
await hooks["chat.params"]!(makeHookInput({ apiId: id }), out)
|
||||
expect(out.maxOutputTokens).toBeUndefined()
|
||||
}
|
||||
})
|
||||
|
||||
test("keeps maxOutputTokens for gpt-5 on non-openai-compatible SDKs", async () => {
|
||||
const hooks = await DropMaxTokensForGpt5Plugin(pluginInput)
|
||||
const out = makeHookOutput()
|
||||
await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-5", npm: "@ai-sdk/openai" }), out)
|
||||
expect(out.maxOutputTokens).toBe(32_000)
|
||||
})
|
||||
|
||||
test("keeps maxOutputTokens for non-gpt-5 models on openai-compatible", async () => {
|
||||
const hooks = await DropMaxTokensForGpt5Plugin(pluginInput)
|
||||
const out = makeHookOutput()
|
||||
await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-4-turbo" }), out)
|
||||
expect(out.maxOutputTokens).toBe(32_000)
|
||||
})
|
||||
Reference in New Issue
Block a user