From de9f11e3990a818ff6d7184f5ea85ee1409a475f Mon Sep 17 00:00:00 2001 From: "kiloconnect[bot]" <240665456+kiloconnect[bot]@users.noreply.github.com> Date: Thu, 7 May 2026 10:36:33 +0000 Subject: [PATCH 1/3] fix(cli): drop max_tokens for gpt-5 models on @ai-sdk/openai-compatible gpt-5 models rejected via openai-compatible proxies (e.g. LiteLLM) because the SDK emits max_tokens while OpenAI requires max_completion_tokens and the compatible SDK cannot rename the field. Clear the cap so the upstream default output budget applies. --- .../drop-max-tokens-gpt5-openai-compatible.md | 5 ++ .../src/kilocode/plugin/drop-max-tokens.ts | 17 +++++ packages/opencode/src/plugin/index.ts | 2 + .../kilocode/drop-max-tokens-gpt5.test.ts | 72 +++++++++++++++++++ 4 files changed, 96 insertions(+) create mode 100644 .changeset/drop-max-tokens-gpt5-openai-compatible.md create mode 100644 packages/opencode/src/kilocode/plugin/drop-max-tokens.ts create mode 100644 packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts diff --git a/.changeset/drop-max-tokens-gpt5-openai-compatible.md b/.changeset/drop-max-tokens-gpt5-openai-compatible.md new file mode 100644 index 00000000000..1bfdbdfa9b3 --- /dev/null +++ b/.changeset/drop-max-tokens-gpt5-openai-compatible.md @@ -0,0 +1,5 @@ +--- +"@kilocode/cli": patch +--- + +Fix gpt-5 models routed through OpenAI-compatible providers (e.g. LiteLLM) rejecting requests with "Unsupported parameter: max_tokens". The CLI now drops the max output token cap for gpt-5 models on `@ai-sdk/openai-compatible`, letting the upstream default output budget apply. diff --git a/packages/opencode/src/kilocode/plugin/drop-max-tokens.ts b/packages/opencode/src/kilocode/plugin/drop-max-tokens.ts new file mode 100644 index 00000000000..404ccd02db0 --- /dev/null +++ b/packages/opencode/src/kilocode/plugin/drop-max-tokens.ts @@ -0,0 +1,17 @@ +import type { Hooks, PluginInput } from "@kilocode/plugin" + +// Drop the max output tokens cap for gpt-5 models routed through +// @ai-sdk/openai-compatible. The SDK always emits `max_tokens`, but OpenAI's +// gpt-5 family (and compatible proxies like LiteLLM) rejects that field and +// requires `max_completion_tokens` instead. The openai-compatible SDK has no +// way to rename the field, so we clear the cap and let the upstream default +// output budget apply. +export async function DropMaxTokensForGpt5Plugin(_input: PluginInput): Promise { + return { + "chat.params": async (input, output) => { + if (input.model.api.npm !== "@ai-sdk/openai-compatible") return + if (!input.model.api.id.toLowerCase().includes("gpt-5")) return + output.maxOutputTokens = undefined + }, + } +} diff --git a/packages/opencode/src/plugin/index.ts b/packages/opencode/src/plugin/index.ts index e944b2f4f95..e796264fb44 100644 --- a/packages/opencode/src/plugin/index.ts +++ b/packages/opencode/src/plugin/index.ts @@ -25,6 +25,7 @@ import { errorMessage } from "@/util/error" import { PluginLoader } from "./loader" import { parsePluginSpecifier, readPluginId, readV1Plugin, resolvePluginId } from "./shared" import { KiloAuthPlugin } from "@kilocode/kilo-gateway" // kilocode_change +import { DropMaxTokensForGpt5Plugin } from "@/kilocode/plugin/drop-max-tokens" // kilocode_change import { registerAdapter } from "@/control-plane/adapters" import type { WorkspaceAdapter } from "@/control-plane/types" @@ -67,6 +68,7 @@ const INTERNAL_PLUGINS: PluginInstance[] = [ CloudflareWorkersAuthPlugin, CloudflareAIGatewayAuthPlugin, AzureAuthPlugin, + DropMaxTokensForGpt5Plugin, ] // kilocode_change end diff --git a/packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts b/packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts new file mode 100644 index 00000000000..9c6c677b55b --- /dev/null +++ b/packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts @@ -0,0 +1,72 @@ +import { expect, test } from "bun:test" +import { DropMaxTokensForGpt5Plugin } from "@/kilocode/plugin/drop-max-tokens" + +const pluginInput = { + client: {} as never, + project: {} as never, + directory: "", + worktree: "", + experimental_workspace: { register() {} }, + serverUrl: new URL("https://example.com"), + $: {} as never, +} + +function makeHookInput(overrides: { providerID?: string; apiId?: string; npm?: string; reasoning?: boolean }) { + return { + sessionID: "s", + agent: "a", + provider: {} as never, + message: {} as never, + model: { + providerID: overrides.providerID ?? "litellm", + api: { + id: overrides.apiId ?? "gpt-5", + url: "", + npm: overrides.npm ?? "@ai-sdk/openai-compatible", + }, + capabilities: { + reasoning: overrides.reasoning ?? true, + temperature: false, + attachment: true, + toolcall: true, + input: { text: true, audio: false, image: false, video: false, pdf: false }, + output: { text: true, audio: false, image: false, video: false, pdf: false }, + interleaved: false, + }, + } as never, + } +} + +function makeHookOutput() { + return { temperature: 0, topP: 1, topK: 0, maxOutputTokens: 32_000 as number | undefined, options: {} } +} + +test("drops maxOutputTokens for gpt-5 on @ai-sdk/openai-compatible", async () => { + const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) + const out = makeHookOutput() + await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-5" }), out) + expect(out.maxOutputTokens).toBeUndefined() +}) + +test("drops maxOutputTokens for gpt-5 variants (codex, pro, mini) on openai-compatible", async () => { + const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) + for (const id of ["gpt-5.2-codex", "gpt-5-pro", "gpt-5-mini", "openai/gpt-5-turbo"]) { + const out = makeHookOutput() + await hooks["chat.params"]!(makeHookInput({ apiId: id }), out) + expect(out.maxOutputTokens).toBeUndefined() + } +}) + +test("keeps maxOutputTokens for gpt-5 on non-openai-compatible SDKs", async () => { + const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) + const out = makeHookOutput() + await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-5", npm: "@ai-sdk/openai" }), out) + expect(out.maxOutputTokens).toBe(32_000) +}) + +test("keeps maxOutputTokens for non-gpt-5 models on openai-compatible", async () => { + const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) + const out = makeHookOutput() + await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-4-turbo" }), out) + expect(out.maxOutputTokens).toBe(32_000) +}) From 8c1b7d715550b188aa1868bc68cabd334db9f961 Mon Sep 17 00:00:00 2001 From: "kiloconnect[bot]" <240665456+kiloconnect[bot]@users.noreply.github.com> Date: Thu, 7 May 2026 11:53:02 +0000 Subject: [PATCH 2/3] refactor(cli): inline gpt-5 max_tokens drop instead of internal plugin Replaces the plugin approach with a narrow conditional at the single call site in session/llm.ts where maxOutputTokens is passed to chat.params. No new file, no new test file, one kilocode_change block. --- .../src/kilocode/plugin/drop-max-tokens.ts | 17 ----- packages/opencode/src/plugin/index.ts | 2 - packages/opencode/src/session/llm.ts | 9 ++- .../kilocode/drop-max-tokens-gpt5.test.ts | 72 ------------------- 4 files changed, 8 insertions(+), 92 deletions(-) delete mode 100644 packages/opencode/src/kilocode/plugin/drop-max-tokens.ts delete mode 100644 packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts diff --git a/packages/opencode/src/kilocode/plugin/drop-max-tokens.ts b/packages/opencode/src/kilocode/plugin/drop-max-tokens.ts deleted file mode 100644 index 404ccd02db0..00000000000 --- a/packages/opencode/src/kilocode/plugin/drop-max-tokens.ts +++ /dev/null @@ -1,17 +0,0 @@ -import type { Hooks, PluginInput } from "@kilocode/plugin" - -// Drop the max output tokens cap for gpt-5 models routed through -// @ai-sdk/openai-compatible. The SDK always emits `max_tokens`, but OpenAI's -// gpt-5 family (and compatible proxies like LiteLLM) rejects that field and -// requires `max_completion_tokens` instead. The openai-compatible SDK has no -// way to rename the field, so we clear the cap and let the upstream default -// output budget apply. -export async function DropMaxTokensForGpt5Plugin(_input: PluginInput): Promise { - return { - "chat.params": async (input, output) => { - if (input.model.api.npm !== "@ai-sdk/openai-compatible") return - if (!input.model.api.id.toLowerCase().includes("gpt-5")) return - output.maxOutputTokens = undefined - }, - } -} diff --git a/packages/opencode/src/plugin/index.ts b/packages/opencode/src/plugin/index.ts index e796264fb44..e944b2f4f95 100644 --- a/packages/opencode/src/plugin/index.ts +++ b/packages/opencode/src/plugin/index.ts @@ -25,7 +25,6 @@ import { errorMessage } from "@/util/error" import { PluginLoader } from "./loader" import { parsePluginSpecifier, readPluginId, readV1Plugin, resolvePluginId } from "./shared" import { KiloAuthPlugin } from "@kilocode/kilo-gateway" // kilocode_change -import { DropMaxTokensForGpt5Plugin } from "@/kilocode/plugin/drop-max-tokens" // kilocode_change import { registerAdapter } from "@/control-plane/adapters" import type { WorkspaceAdapter } from "@/control-plane/types" @@ -68,7 +67,6 @@ const INTERNAL_PLUGINS: PluginInstance[] = [ CloudflareWorkersAuthPlugin, CloudflareAIGatewayAuthPlugin, AzureAuthPlugin, - DropMaxTokensForGpt5Plugin, ] // kilocode_change end diff --git a/packages/opencode/src/session/llm.ts b/packages/opencode/src/session/llm.ts index 39164727d12..7cdb5e85c05 100644 --- a/packages/opencode/src/session/llm.ts +++ b/packages/opencode/src/session/llm.ts @@ -186,7 +186,14 @@ const live: Layer.Layer< : undefined, topP: input.agent.topP ?? ProviderTransform.topP(input.model), topK: ProviderTransform.topK(input.model), - maxOutputTokens: ProviderTransform.maxOutputTokens(input.model), + // kilocode_change start - gpt-5 via @ai-sdk/openai-compatible proxies (e.g. LiteLLM) + // rejects `max_tokens`; OpenAI requires `max_completion_tokens` and the compatible + // SDK cannot rename the field, so drop the cap and let the upstream default apply. + maxOutputTokens: + input.model.api.npm === "@ai-sdk/openai-compatible" && input.model.api.id.toLowerCase().includes("gpt-5") + ? undefined + : ProviderTransform.maxOutputTokens(input.model), + // kilocode_change end options, }, ) diff --git a/packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts b/packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts deleted file mode 100644 index 9c6c677b55b..00000000000 --- a/packages/opencode/test/kilocode/drop-max-tokens-gpt5.test.ts +++ /dev/null @@ -1,72 +0,0 @@ -import { expect, test } from "bun:test" -import { DropMaxTokensForGpt5Plugin } from "@/kilocode/plugin/drop-max-tokens" - -const pluginInput = { - client: {} as never, - project: {} as never, - directory: "", - worktree: "", - experimental_workspace: { register() {} }, - serverUrl: new URL("https://example.com"), - $: {} as never, -} - -function makeHookInput(overrides: { providerID?: string; apiId?: string; npm?: string; reasoning?: boolean }) { - return { - sessionID: "s", - agent: "a", - provider: {} as never, - message: {} as never, - model: { - providerID: overrides.providerID ?? "litellm", - api: { - id: overrides.apiId ?? "gpt-5", - url: "", - npm: overrides.npm ?? "@ai-sdk/openai-compatible", - }, - capabilities: { - reasoning: overrides.reasoning ?? true, - temperature: false, - attachment: true, - toolcall: true, - input: { text: true, audio: false, image: false, video: false, pdf: false }, - output: { text: true, audio: false, image: false, video: false, pdf: false }, - interleaved: false, - }, - } as never, - } -} - -function makeHookOutput() { - return { temperature: 0, topP: 1, topK: 0, maxOutputTokens: 32_000 as number | undefined, options: {} } -} - -test("drops maxOutputTokens for gpt-5 on @ai-sdk/openai-compatible", async () => { - const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) - const out = makeHookOutput() - await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-5" }), out) - expect(out.maxOutputTokens).toBeUndefined() -}) - -test("drops maxOutputTokens for gpt-5 variants (codex, pro, mini) on openai-compatible", async () => { - const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) - for (const id of ["gpt-5.2-codex", "gpt-5-pro", "gpt-5-mini", "openai/gpt-5-turbo"]) { - const out = makeHookOutput() - await hooks["chat.params"]!(makeHookInput({ apiId: id }), out) - expect(out.maxOutputTokens).toBeUndefined() - } -}) - -test("keeps maxOutputTokens for gpt-5 on non-openai-compatible SDKs", async () => { - const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) - const out = makeHookOutput() - await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-5", npm: "@ai-sdk/openai" }), out) - expect(out.maxOutputTokens).toBe(32_000) -}) - -test("keeps maxOutputTokens for non-gpt-5 models on openai-compatible", async () => { - const hooks = await DropMaxTokensForGpt5Plugin(pluginInput) - const out = makeHookOutput() - await hooks["chat.params"]!(makeHookInput({ apiId: "gpt-4-turbo" }), out) - expect(out.maxOutputTokens).toBe(32_000) -}) From 18f435d5731a5a843013794be07ce5b0d7591e3b Mon Sep 17 00:00:00 2001 From: "kiloconnect[bot]" <240665456+kiloconnect[bot]@users.noreply.github.com> Date: Thu, 7 May 2026 12:00:46 +0000 Subject: [PATCH 3/3] docs: tighten changeset for end-user release notes --- .changeset/drop-max-tokens-gpt5-openai-compatible.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.changeset/drop-max-tokens-gpt5-openai-compatible.md b/.changeset/drop-max-tokens-gpt5-openai-compatible.md index 1bfdbdfa9b3..cdf54e8d48d 100644 --- a/.changeset/drop-max-tokens-gpt5-openai-compatible.md +++ b/.changeset/drop-max-tokens-gpt5-openai-compatible.md @@ -2,4 +2,4 @@ "@kilocode/cli": patch --- -Fix gpt-5 models routed through OpenAI-compatible providers (e.g. LiteLLM) rejecting requests with "Unsupported parameter: max_tokens". The CLI now drops the max output token cap for gpt-5 models on `@ai-sdk/openai-compatible`, letting the upstream default output budget apply. +Fix gpt-5 models failing with `Unsupported parameter: max_tokens` when accessed through custom OpenAI-compatible providers such as LiteLLM.