From ddeb19790a480dc3e3998da7d05a64b9295042bc Mon Sep 17 00:00:00 2001 From: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:10:57 -0500 Subject: [PATCH] fix(ai): replay Kimi reasoning details without the streaming index (#50383) --- packages/ai/src/protocols/openai-chat.ts | 149 +++++++++++--- packages/ai/test/provider/openai-chat.test.ts | 187 +++++++++++++++++- packages/ai/test/provider/openrouter.test.ts | 13 +- 3 files changed, 314 insertions(+), 35 deletions(-) diff --git a/packages/ai/src/protocols/openai-chat.ts b/packages/ai/src/protocols/openai-chat.ts index 2320f45dcc1..1dcc0ef69a2 100644 --- a/packages/ai/src/protocols/openai-chat.ts +++ b/packages/ai/src/protocols/openai-chat.ts @@ -1,4 +1,4 @@ -import { Effect, Schema } from "effect" +import { Effect, Option, Schema } from "effect" import { Tool } from "@opencode/schema/tool" import { Route } from "../route/client.js" import { Auth } from "../route/auth.js" @@ -76,6 +76,44 @@ const OpenAIChatAssistantToolCall = Schema.Struct({ }) type OpenAIChatAssistantToolCall = Schema.Schema.Type +// `reasoning_details` carries two dialects. OpenRouter's `reasoning.*` entries +// must be replayed unmodified (`index` included), so they keep every field they +// arrived with. Kimi's OpenAI-compatible surface streams preserved thinking as +// bare `summary` / `encrypted` entries keyed by a stream-only `index`; Kimi does +// not document this publicly, so the handling follows Kimi Code (Kimi's own +// client): merge summary deltas by `index`, replay without `index`, and always +// send `reasoning_content` alongside. Anything else is dropped at the boundary. +const OpenRouterDetailFields = { + id: Schema.optional(Schema.NullOr(Schema.String)), + format: Schema.optional(Schema.String), + index: Schema.optional(Schema.Number), + signature: Schema.optional(Schema.NullOr(Schema.String)), +} +const ReasoningDetail = Schema.Union([ + Schema.StructWithRest( + Schema.Struct({ type: Schema.Literal("reasoning.text"), text: Schema.optional(Schema.String), ...OpenRouterDetailFields }), + [Schema.Record(Schema.String, Schema.Unknown)], + ), + Schema.StructWithRest( + Schema.Struct({ + type: Schema.Literal("reasoning.summary"), + summary: Schema.optional(Schema.String), + ...OpenRouterDetailFields, + }), + [Schema.Record(Schema.String, Schema.Unknown)], + ), + Schema.StructWithRest( + Schema.Struct({ type: Schema.Literal("reasoning.encrypted"), data: Schema.String, ...OpenRouterDetailFields }), + [Schema.Record(Schema.String, Schema.Unknown)], + ), + Schema.Struct({ type: Schema.Literal("summary"), summary: Schema.String, index: Schema.optional(Schema.Number) }), + Schema.Struct({ type: Schema.Literal("encrypted"), encrypted: Schema.String, index: Schema.optional(Schema.Number) }), +]) +type ReasoningDetail = Schema.Schema.Type +const decodeReasoningDetail = Schema.decodeUnknownOption(ReasoningDetail) +const knownReasoningDetails = (details: ReadonlyArray) => + details.flatMap((detail) => Option.toArray(decodeReasoningDetail(detail))) + // Intentionally omit Gemini's provider-specific `extra_content.google.thought_signature` // extension until direct Google OpenAI-compatible routing is supported here: // https://github.com/vercel/ai/issues/11590 @@ -265,7 +303,9 @@ export interface ParserState { readonly finishReason?: FinishReasonDetails readonly lifecycle: Lifecycle.State readonly reasoningField?: string - readonly reasoningDetails: Array + /** A scalar reasoning field (`reasoning_content`, ...) has carried text in this stream. */ + readonly reasoningTextObserved: boolean + readonly reasoningDetails: Array readonly reasoningDetailsObserved: boolean readonly reasoningEmitted: boolean readonly latestToolIndex?: number @@ -341,10 +381,21 @@ const reasoningDetails = (parts: ReadonlyArray, native: unknown, return Array.isArray(details) ? details : [] }) if (parts.some((part) => Array.isArray(part.providerMetadata?.[providerMetadataKey]?.reasoningDetails))) - return observed - if (isRecord(native) && Array.isArray(native.reasoning_details)) return native.reasoning_details + return knownReasoningDetails(observed).map(lowerReasoningDetail) + if (isRecord(native) && Array.isArray(native.reasoning_details)) + return knownReasoningDetails(native.reasoning_details).map(lowerReasoningDetail) } +// Kimi rejects its stream-only `index` on requests +// ("the reasoning_details ... must not contain streaming index"). +const lowerReasoningDetail = (detail: ReasoningDetail) => { + if (detail.type === "summary") return { type: detail.type, summary: detail.summary } + if (detail.type === "encrypted") return { type: detail.type, encrypted: detail.encrypted } + return detail +} + +const isKimiDetail = (detail: { readonly type: string }) => detail.type === "summary" || detail.type === "encrypted" + const lowerUserMessage = Effect.fn("OpenAIChat.lowerUserMessage")(function* ( message: OpenAIChatRequestMessage, options: LoweringOptions, @@ -410,6 +461,9 @@ const lowerAssistantMessage = Effect.fn("OpenAIChat.lowerAssistantMessage")(func if (observedField !== undefined) return observedField if (nativeReasoning !== undefined) return "reasoning_content" if (!fullyStructured || requireReasoning) return "reasoning_content" + // Kimi always expects `reasoning_content` on replayed assistant messages, + // even when thinking arrived only through structured details. + if (details?.some(isKimiDetail)) return "reasoning_content" })() const reasoningText = (() => { if (configuredField !== undefined) @@ -881,44 +935,74 @@ const reasoningDelta = ( return undefined } -const detailText = (details: ReadonlyArray) => { +const detailText = (details: ReadonlyArray, hideKimiSummary: boolean) => { const text = details.flatMap((detail) => { - if (!isRecord(detail)) return [] - if (detail.type === "reasoning.text" && typeof detail.text === "string" && detail.text) return [detail.text] - if (detail.type === "reasoning.summary" && typeof detail.summary === "string" && detail.summary) - return [detail.summary] + if (detail.type === "reasoning.text") return detail.text ? [detail.text] : [] + if (detail.type === "reasoning.summary") return detail.summary ? [detail.summary] : [] + // Kimi streams the full thinking through `reasoning_content` and a separate + // summary through details; show the summary only when nothing else does. + if (detail.type === "summary") return detail.summary && !hideKimiSummary ? [detail.summary] : [] return [] }) if (text.length > 0) return text.join("") } -const appendReasoningDetails = (result: Array, details: ReadonlyArray) => { +const appendReasoningDetails = (result: Array, details: ReadonlyArray) => { for (const detail of details) { const previous = result.at(-1) - if ( - !isRecord(previous) || - previous.type !== "reasoning.text" || - !isRecord(detail) || - detail.type !== "reasoning.text" || - conflictingReasoningTextDetails(previous, detail) - ) { + const merged = previous === undefined ? undefined : mergeReasoningDetails(previous, detail) + if (merged === undefined) { result.push(detail) continue } - result[result.length - 1] = { - ...previous, - ...Object.fromEntries(Object.entries(detail).filter((entry) => entry[1] !== undefined)), - text: `${typeof previous.text === "string" ? previous.text : ""}${typeof detail.text === "string" ? detail.text : ""}`, - signature: mergeDetailValue(previous.signature, detail.signature), - format: mergeDetailValue(previous.format, detail.format), - } + result[result.length - 1] = merged } } -const mergeDetailValue = (previous: unknown, current: unknown) => +// Consecutive text or summary deltas of the same kind accumulate into one +// entry; encrypted entries are opaque and never merge. +const mergeReasoningDetails = (previous: ReasoningDetail, detail: ReasoningDetail): ReasoningDetail | undefined => { + if (conflictingReasoningDetails(previous, detail)) return undefined + if (previous.type === "reasoning.text" && detail.type === "reasoning.text") + return { + ...previous, + ...detail, + text: `${previous.text ?? ""}${detail.text ?? ""}`, + ...mergeDetailIdentity(previous, detail), + } + if (previous.type === "reasoning.summary" && detail.type === "reasoning.summary") + return { + ...previous, + ...detail, + summary: `${previous.summary ?? ""}${detail.summary ?? ""}`, + ...mergeDetailIdentity(previous, detail), + } + if (previous.type === "summary" && detail.type === "summary") + return { ...previous, ...detail, summary: previous.summary + detail.summary } +} + +type DetailIdentity = { + readonly id?: string | null + readonly index?: number + readonly format?: string + readonly signature?: string | null +} + +// The first non-empty signature and format win; a later delta may carry the +// signature for text that streamed earlier. +const mergeDetailIdentity = (previous: DetailIdentity, current: DetailIdentity) => { + const signature = mergeDetailValue(previous.signature, current.signature) + const format = mergeDetailValue(previous.format, current.format) + return { + ...(signature === undefined ? {} : { signature }), + ...(format === undefined ? {} : { format }), + } +} + +const mergeDetailValue = (previous: T | undefined, current: T | undefined) => previous || current || (previous !== undefined ? previous : current) -const conflictingReasoningTextDetails = (previous: Record, current: Record) => +const conflictingReasoningDetails = (previous: DetailIdentity, current: DetailIdentity) => conflictingDetailValue(previous.id, current.id) || conflictingDetailValue(previous.index, current.index) || conflictingDetailValue(previous.format, current.format) || @@ -930,7 +1014,7 @@ const conflictingDetailValue = (previous: unknown, current: unknown) => const reasoningMetadata = ( providerMetadataKey: string, field: ParserState["reasoningField"], - details?: ReadonlyArray, + details?: ReadonlyArray, ) => ({ [providerMetadataKey]: { ...(field ? { reasoningField: field } : {}), @@ -993,11 +1077,16 @@ const step = (state: ParserState, event: OpenAIChatEvent) => } const reasoningField = state.reasoningField ?? reasoning?.field - const detailDelta = Array.isArray(delta?.reasoning_details) ? delta.reasoning_details : undefined + const reasoningTextObserved = state.reasoningTextObserved || reasoning !== undefined + const detailDelta = Array.isArray(delta?.reasoning_details) + ? knownReasoningDetails(delta.reasoning_details) + : undefined if (detailDelta !== undefined) appendReasoningDetails(state.reasoningDetails, detailDelta) const reasoningDetailsObserved = state.reasoningDetailsObserved || detailDelta !== undefined const deltaMetadata = reasoningMetadata(state.providerMetadataKey, reasoningField) - const text = detailDelta?.length ? (detailText(detailDelta) ?? reasoning?.text) : reasoning?.text + const text = detailDelta?.length + ? (detailText(detailDelta, reasoningTextObserved) ?? reasoning?.text) + : reasoning?.text if (text !== undefined) lifecycle = Lifecycle.reasoningDelta(lifecycle, events, "reasoning-0", text, deltaMetadata) else if ( reasoningDetailsObserved && @@ -1093,6 +1182,7 @@ const step = (state: ParserState, event: OpenAIChatEvent) => finishReason, lifecycle, reasoningField, + reasoningTextObserved, reasoningDetails: state.reasoningDetails, reasoningDetailsObserved, reasoningEmitted, @@ -1173,6 +1263,7 @@ export const protocol = Protocol.make({ toolCallEvents: [], lifecycle: Lifecycle.initial(), reasoningField: request.model.compatibility?.reasoningField, + reasoningTextObserved: false, reasoningDetails: [], reasoningDetailsObserved: false, reasoningEmitted: false, diff --git a/packages/ai/test/provider/openai-chat.test.ts b/packages/ai/test/provider/openai-chat.test.ts index b1f5a1858d0..453b37af6a7 100644 --- a/packages/ai/test/provider/openai-chat.test.ts +++ b/packages/ai/test/provider/openai-chat.test.ts @@ -1182,7 +1182,9 @@ describe("OpenAI Chat route", () => { }), ) - it.effect("preserves unknown reasoning details while using scalar display text", () => + // Only recognized detail shapes are retained and replayed; echoing an + // undocumented provider payload is what breaks follow-up requests. + it.effect("drops unknown reasoning details while using scalar display text", () => Effect.gen(function* () { const details = [{ type: "reasoning.future", format: "provider-v2", state: { opaque: true } }] const response = yield* LLMClient.generate(request).pipe( @@ -1199,16 +1201,195 @@ describe("OpenAI Chat route", () => { expect(response.reasoning).toBe("thinking") expect(response.message.content.find((part) => part.type === "reasoning")?.providerMetadata).toEqual({ - openai: { reasoningField: "reasoning", reasoningDetails: details }, + openai: { reasoningField: "reasoning", reasoningDetails: [] }, }) const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] })) expect(replay.body.messages).toEqual([ - { role: "assistant", content: "Hello", reasoning: "thinking", reasoning_details: details }, + { role: "assistant", content: "Hello", reasoning: "thinking", reasoning_details: [] }, ]) }), ) + // Kimi's coding endpoint streams the full thinking through `reasoning_content` + // and a separate summary + encrypted blob through its own `reasoning_details` + // dialect. The stream-only `index` must not be echoed back. + it.effect("merges Kimi summary deltas by index and replays details without the streaming index", () => + Effect.gen(function* () { + const response = yield* LLMClient.generate(request).pipe( + Effect.provide( + fixedResponse( + sseEvents( + { choices: [{ delta: { reasoning_content: "Let me" } }] }, + { choices: [{ delta: { reasoning_content: " think" } }] }, + { choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: "Plan" }] } }] }, + { choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: " tools" }] } }] }, + { choices: [{ delta: { reasoning_details: [{ index: 1, type: "encrypted", encrypted: "opaque" }] } }] }, + { + choices: [ + { + delta: { + tool_calls: [ + { index: 0, id: "call_1", type: "function", function: { name: "get_time", arguments: "{}" } }, + ], + }, + }, + ], + }, + { choices: [{ delta: {}, finish_reason: "tool_calls" }] }, + ), + ), + ), + ) + + const stored = [ + { index: 0, type: "summary", summary: "Plan tools" }, + { index: 1, type: "encrypted", encrypted: "opaque" }, + ] + expect(response.reasoning).toBe("Let me think") + expect(response.message.content.find((part) => part.type === "reasoning")?.providerMetadata).toEqual({ + openai: { reasoningField: "reasoning_content", reasoningDetails: stored }, + }) + + const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] })) + expect(replay.body.messages).toEqual([ + { + role: "assistant", + content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "get_time", arguments: "{}" } }], + reasoning_content: "Let me think", + reasoning_details: [ + { type: "summary", summary: "Plan tools" }, + { type: "encrypted", encrypted: "opaque" }, + ], + }, + ]) + }), + ) + + it.effect("displays Kimi summaries and replays reasoning_content when no scalar reasoning streams", () => + Effect.gen(function* () { + const response = yield* LLMClient.generate(request).pipe( + Effect.provide( + fixedResponse( + sseEvents( + { choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: "Plan" }] } }] }, + { choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: " tools" }] } }] }, + { choices: [{ delta: { reasoning_details: [{ index: 1, type: "encrypted", encrypted: "opaque" }] } }] }, + { choices: [{ delta: { content: "Hello" } }] }, + { choices: [{ delta: {}, finish_reason: "stop" }] }, + ), + ), + ), + ) + + expect(response.reasoning).toBe("Plan tools") + + const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] })) + expect(replay.body.messages).toEqual([ + { + role: "assistant", + content: "Hello", + reasoning_content: "Plan tools", + reasoning_details: [ + { type: "summary", summary: "Plan tools" }, + { type: "encrypted", encrypted: "opaque" }, + ], + }, + ]) + }), + ) + + // Sessions persisted before the fix already hold Kimi details with `index`. + it.effect("strips the streaming index from previously stored Kimi details", () => + Effect.gen(function* () { + const replay = yield* compileRequest( + LLM.request({ + model, + messages: [ + Message.assistant([ + { + type: "reasoning", + text: "thinking", + providerMetadata: { + openai: { + reasoningField: "reasoning_content", + reasoningDetails: [ + { index: 0, type: "summary", summary: "thinking" }, + { index: 1, type: "encrypted", encrypted: "opaque" }, + ], + }, + }, + }, + ]), + ], + }), + ) + + expect(replay.body.messages).toEqual([ + { + role: "assistant", + content: "", + reasoning_content: "thinking", + reasoning_details: [ + { type: "summary", summary: "thinking" }, + { type: "encrypted", encrypted: "opaque" }, + ], + }, + ]) + }), + ) + + it.effect("merges consecutive OpenRouter summary deltas and replays them unmodified", () => + Effect.gen(function* () { + const merged = [ + { type: "reasoning.summary", summary: "Plan tools", format: "openai-responses-v1", index: 0 }, + { type: "reasoning.encrypted", data: "opaque", format: "openai-responses-v1", index: 0 }, + ] + const response = yield* LLMClient.generate(request).pipe( + Effect.provide( + fixedResponse( + sseEvents( + { + choices: [ + { + delta: { + reasoning_details: [ + { type: "reasoning.summary", summary: "Plan", format: "openai-responses-v1", index: 0 }, + ], + }, + }, + ], + }, + { + choices: [ + { + delta: { + reasoning_details: [ + { type: "reasoning.summary", summary: " tools", format: "openai-responses-v1", index: 0 }, + ], + }, + }, + ], + }, + { choices: [{ delta: { reasoning_details: [merged[1]] } }] }, + { choices: [{ delta: { content: "Hello" } }] }, + { choices: [{ delta: {}, finish_reason: "stop" }] }, + ), + ), + ), + ) + + expect(response.reasoning).toBe("Plan tools") + expect(response.message.content.find((part) => part.type === "reasoning")?.providerMetadata).toEqual({ + openai: { reasoningDetails: merged }, + }) + + const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] })) + expect(replay.body.messages).toEqual([{ role: "assistant", content: "Hello", reasoning_details: merged }]) + }), + ) + it.effect("uses scalar display text for signature-only reasoning details", () => Effect.gen(function* () { const details = [{ type: "reasoning.text", signature: "signed", format: "provider-v2", index: 0 }] diff --git a/packages/ai/test/provider/openrouter.test.ts b/packages/ai/test/provider/openrouter.test.ts index b2ae31413ce..7c320df3420 100644 --- a/packages/ai/test/provider/openrouter.test.ts +++ b/packages/ai/test/provider/openrouter.test.ts @@ -315,10 +315,9 @@ describe("OpenRouter", () => { }), ) - it.effect("preserves opaque and duplicate continuation details", () => + it.effect("drops unrecognized details and preserves duplicate continuation details", () => Effect.gen(function* () { const details = [ - { type: "reasoning.future", format: "provider-v2", state: { opaque: true } }, { type: "reasoning.encrypted", id: "state", data: "opaque" }, { type: "reasoning.encrypted", id: "state", data: "opaque" }, ] @@ -330,7 +329,15 @@ describe("OpenRouter", () => { Message.assistant({ type: "reasoning", text: "Thinking", - providerMetadata: { openrouter: { reasoningField: "reasoning", reasoningDetails: details } }, + providerMetadata: { + openrouter: { + reasoningField: "reasoning", + reasoningDetails: [ + { type: "reasoning.future", format: "provider-v2", state: { opaque: true } }, + ...details, + ], + }, + }, }), ], }),