fix(ai): replay Kimi reasoning details without the streaming index (#50383)

This commit is contained in:
Aiden Cline
2026-09-22 19:10:57 -05:00
committed by GitHub
parent fe0d1682ca
commit ddeb19790a
3 changed files with 314 additions and 35 deletions
+120 -29
View File
@@ -1,4 +1,4 @@
import { Effect, Schema } from "effect"
import { Effect, Option, Schema } from "effect"
import { Tool } from "@opencode/schema/tool"
import { Route } from "../route/client.js"
import { Auth } from "../route/auth.js"
@@ -76,6 +76,44 @@ const OpenAIChatAssistantToolCall = Schema.Struct({
})
type OpenAIChatAssistantToolCall = Schema.Schema.Type<typeof OpenAIChatAssistantToolCall>
// `reasoning_details` carries two dialects. OpenRouter's `reasoning.*` entries
// must be replayed unmodified (`index` included), so they keep every field they
// arrived with. Kimi's OpenAI-compatible surface streams preserved thinking as
// bare `summary` / `encrypted` entries keyed by a stream-only `index`; Kimi does
// not document this publicly, so the handling follows Kimi Code (Kimi's own
// client): merge summary deltas by `index`, replay without `index`, and always
// send `reasoning_content` alongside. Anything else is dropped at the boundary.
const OpenRouterDetailFields = {
id: Schema.optional(Schema.NullOr(Schema.String)),
format: Schema.optional(Schema.String),
index: Schema.optional(Schema.Number),
signature: Schema.optional(Schema.NullOr(Schema.String)),
}
const ReasoningDetail = Schema.Union([
Schema.StructWithRest(
Schema.Struct({ type: Schema.Literal("reasoning.text"), text: Schema.optional(Schema.String), ...OpenRouterDetailFields }),
[Schema.Record(Schema.String, Schema.Unknown)],
),
Schema.StructWithRest(
Schema.Struct({
type: Schema.Literal("reasoning.summary"),
summary: Schema.optional(Schema.String),
...OpenRouterDetailFields,
}),
[Schema.Record(Schema.String, Schema.Unknown)],
),
Schema.StructWithRest(
Schema.Struct({ type: Schema.Literal("reasoning.encrypted"), data: Schema.String, ...OpenRouterDetailFields }),
[Schema.Record(Schema.String, Schema.Unknown)],
),
Schema.Struct({ type: Schema.Literal("summary"), summary: Schema.String, index: Schema.optional(Schema.Number) }),
Schema.Struct({ type: Schema.Literal("encrypted"), encrypted: Schema.String, index: Schema.optional(Schema.Number) }),
])
type ReasoningDetail = Schema.Schema.Type<typeof ReasoningDetail>
const decodeReasoningDetail = Schema.decodeUnknownOption(ReasoningDetail)
const knownReasoningDetails = (details: ReadonlyArray<unknown>) =>
details.flatMap((detail) => Option.toArray(decodeReasoningDetail(detail)))
// Intentionally omit Gemini's provider-specific `extra_content.google.thought_signature`
// extension until direct Google OpenAI-compatible routing is supported here:
// https://github.com/vercel/ai/issues/11590
@@ -265,7 +303,9 @@ export interface ParserState {
readonly finishReason?: FinishReasonDetails
readonly lifecycle: Lifecycle.State
readonly reasoningField?: string
readonly reasoningDetails: Array<unknown>
/** A scalar reasoning field (`reasoning_content`, ...) has carried text in this stream. */
readonly reasoningTextObserved: boolean
readonly reasoningDetails: Array<ReasoningDetail>
readonly reasoningDetailsObserved: boolean
readonly reasoningEmitted: boolean
readonly latestToolIndex?: number
@@ -341,10 +381,21 @@ const reasoningDetails = (parts: ReadonlyArray<ReasoningPart>, native: unknown,
return Array.isArray(details) ? details : []
})
if (parts.some((part) => Array.isArray(part.providerMetadata?.[providerMetadataKey]?.reasoningDetails)))
return observed
if (isRecord(native) && Array.isArray(native.reasoning_details)) return native.reasoning_details
return knownReasoningDetails(observed).map(lowerReasoningDetail)
if (isRecord(native) && Array.isArray(native.reasoning_details))
return knownReasoningDetails(native.reasoning_details).map(lowerReasoningDetail)
}
// Kimi rejects its stream-only `index` on requests
// ("the reasoning_details ... must not contain streaming index").
const lowerReasoningDetail = (detail: ReasoningDetail) => {
if (detail.type === "summary") return { type: detail.type, summary: detail.summary }
if (detail.type === "encrypted") return { type: detail.type, encrypted: detail.encrypted }
return detail
}
const isKimiDetail = (detail: { readonly type: string }) => detail.type === "summary" || detail.type === "encrypted"
const lowerUserMessage = Effect.fn("OpenAIChat.lowerUserMessage")(function* (
message: OpenAIChatRequestMessage,
options: LoweringOptions,
@@ -410,6 +461,9 @@ const lowerAssistantMessage = Effect.fn("OpenAIChat.lowerAssistantMessage")(func
if (observedField !== undefined) return observedField
if (nativeReasoning !== undefined) return "reasoning_content"
if (!fullyStructured || requireReasoning) return "reasoning_content"
// Kimi always expects `reasoning_content` on replayed assistant messages,
// even when thinking arrived only through structured details.
if (details?.some(isKimiDetail)) return "reasoning_content"
})()
const reasoningText = (() => {
if (configuredField !== undefined)
@@ -881,44 +935,74 @@ const reasoningDelta = (
return undefined
}
const detailText = (details: ReadonlyArray<unknown>) => {
const detailText = (details: ReadonlyArray<ReasoningDetail>, hideKimiSummary: boolean) => {
const text = details.flatMap((detail) => {
if (!isRecord(detail)) return []
if (detail.type === "reasoning.text" && typeof detail.text === "string" && detail.text) return [detail.text]
if (detail.type === "reasoning.summary" && typeof detail.summary === "string" && detail.summary)
return [detail.summary]
if (detail.type === "reasoning.text") return detail.text ? [detail.text] : []
if (detail.type === "reasoning.summary") return detail.summary ? [detail.summary] : []
// Kimi streams the full thinking through `reasoning_content` and a separate
// summary through details; show the summary only when nothing else does.
if (detail.type === "summary") return detail.summary && !hideKimiSummary ? [detail.summary] : []
return []
})
if (text.length > 0) return text.join("")
}
const appendReasoningDetails = (result: Array<unknown>, details: ReadonlyArray<unknown>) => {
const appendReasoningDetails = (result: Array<ReasoningDetail>, details: ReadonlyArray<ReasoningDetail>) => {
for (const detail of details) {
const previous = result.at(-1)
if (
!isRecord(previous) ||
previous.type !== "reasoning.text" ||
!isRecord(detail) ||
detail.type !== "reasoning.text" ||
conflictingReasoningTextDetails(previous, detail)
) {
const merged = previous === undefined ? undefined : mergeReasoningDetails(previous, detail)
if (merged === undefined) {
result.push(detail)
continue
}
result[result.length - 1] = {
...previous,
...Object.fromEntries(Object.entries(detail).filter((entry) => entry[1] !== undefined)),
text: `${typeof previous.text === "string" ? previous.text : ""}${typeof detail.text === "string" ? detail.text : ""}`,
signature: mergeDetailValue(previous.signature, detail.signature),
format: mergeDetailValue(previous.format, detail.format),
}
result[result.length - 1] = merged
}
}
const mergeDetailValue = (previous: unknown, current: unknown) =>
// Consecutive text or summary deltas of the same kind accumulate into one
// entry; encrypted entries are opaque and never merge.
const mergeReasoningDetails = (previous: ReasoningDetail, detail: ReasoningDetail): ReasoningDetail | undefined => {
if (conflictingReasoningDetails(previous, detail)) return undefined
if (previous.type === "reasoning.text" && detail.type === "reasoning.text")
return {
...previous,
...detail,
text: `${previous.text ?? ""}${detail.text ?? ""}`,
...mergeDetailIdentity(previous, detail),
}
if (previous.type === "reasoning.summary" && detail.type === "reasoning.summary")
return {
...previous,
...detail,
summary: `${previous.summary ?? ""}${detail.summary ?? ""}`,
...mergeDetailIdentity(previous, detail),
}
if (previous.type === "summary" && detail.type === "summary")
return { ...previous, ...detail, summary: previous.summary + detail.summary }
}
type DetailIdentity = {
readonly id?: string | null
readonly index?: number
readonly format?: string
readonly signature?: string | null
}
// The first non-empty signature and format win; a later delta may carry the
// signature for text that streamed earlier.
const mergeDetailIdentity = (previous: DetailIdentity, current: DetailIdentity) => {
const signature = mergeDetailValue(previous.signature, current.signature)
const format = mergeDetailValue(previous.format, current.format)
return {
...(signature === undefined ? {} : { signature }),
...(format === undefined ? {} : { format }),
}
}
const mergeDetailValue = <T>(previous: T | undefined, current: T | undefined) =>
previous || current || (previous !== undefined ? previous : current)
const conflictingReasoningTextDetails = (previous: Record<string, unknown>, current: Record<string, unknown>) =>
const conflictingReasoningDetails = (previous: DetailIdentity, current: DetailIdentity) =>
conflictingDetailValue(previous.id, current.id) ||
conflictingDetailValue(previous.index, current.index) ||
conflictingDetailValue(previous.format, current.format) ||
@@ -930,7 +1014,7 @@ const conflictingDetailValue = (previous: unknown, current: unknown) =>
const reasoningMetadata = (
providerMetadataKey: string,
field: ParserState["reasoningField"],
details?: ReadonlyArray<unknown>,
details?: ReadonlyArray<ReasoningDetail>,
) => ({
[providerMetadataKey]: {
...(field ? { reasoningField: field } : {}),
@@ -993,11 +1077,16 @@ const step = (state: ParserState, event: OpenAIChatEvent) =>
}
const reasoningField = state.reasoningField ?? reasoning?.field
const detailDelta = Array.isArray(delta?.reasoning_details) ? delta.reasoning_details : undefined
const reasoningTextObserved = state.reasoningTextObserved || reasoning !== undefined
const detailDelta = Array.isArray(delta?.reasoning_details)
? knownReasoningDetails(delta.reasoning_details)
: undefined
if (detailDelta !== undefined) appendReasoningDetails(state.reasoningDetails, detailDelta)
const reasoningDetailsObserved = state.reasoningDetailsObserved || detailDelta !== undefined
const deltaMetadata = reasoningMetadata(state.providerMetadataKey, reasoningField)
const text = detailDelta?.length ? (detailText(detailDelta) ?? reasoning?.text) : reasoning?.text
const text = detailDelta?.length
? (detailText(detailDelta, reasoningTextObserved) ?? reasoning?.text)
: reasoning?.text
if (text !== undefined) lifecycle = Lifecycle.reasoningDelta(lifecycle, events, "reasoning-0", text, deltaMetadata)
else if (
reasoningDetailsObserved &&
@@ -1093,6 +1182,7 @@ const step = (state: ParserState, event: OpenAIChatEvent) =>
finishReason,
lifecycle,
reasoningField,
reasoningTextObserved,
reasoningDetails: state.reasoningDetails,
reasoningDetailsObserved,
reasoningEmitted,
@@ -1173,6 +1263,7 @@ export const protocol = Protocol.make({
toolCallEvents: [],
lifecycle: Lifecycle.initial(),
reasoningField: request.model.compatibility?.reasoningField,
reasoningTextObserved: false,
reasoningDetails: [],
reasoningDetailsObserved: false,
reasoningEmitted: false,
+184 -3
View File
@@ -1182,7 +1182,9 @@ describe("OpenAI Chat route", () => {
}),
)
it.effect("preserves unknown reasoning details while using scalar display text", () =>
// Only recognized detail shapes are retained and replayed; echoing an
// undocumented provider payload is what breaks follow-up requests.
it.effect("drops unknown reasoning details while using scalar display text", () =>
Effect.gen(function* () {
const details = [{ type: "reasoning.future", format: "provider-v2", state: { opaque: true } }]
const response = yield* LLMClient.generate(request).pipe(
@@ -1199,16 +1201,195 @@ describe("OpenAI Chat route", () => {
expect(response.reasoning).toBe("thinking")
expect(response.message.content.find((part) => part.type === "reasoning")?.providerMetadata).toEqual({
openai: { reasoningField: "reasoning", reasoningDetails: details },
openai: { reasoningField: "reasoning", reasoningDetails: [] },
})
const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] }))
expect(replay.body.messages).toEqual([
{ role: "assistant", content: "Hello", reasoning: "thinking", reasoning_details: details },
{ role: "assistant", content: "Hello", reasoning: "thinking", reasoning_details: [] },
])
}),
)
// Kimi's coding endpoint streams the full thinking through `reasoning_content`
// and a separate summary + encrypted blob through its own `reasoning_details`
// dialect. The stream-only `index` must not be echoed back.
it.effect("merges Kimi summary deltas by index and replays details without the streaming index", () =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(request).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{ choices: [{ delta: { reasoning_content: "Let me" } }] },
{ choices: [{ delta: { reasoning_content: " think" } }] },
{ choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: "Plan" }] } }] },
{ choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: " tools" }] } }] },
{ choices: [{ delta: { reasoning_details: [{ index: 1, type: "encrypted", encrypted: "opaque" }] } }] },
{
choices: [
{
delta: {
tool_calls: [
{ index: 0, id: "call_1", type: "function", function: { name: "get_time", arguments: "{}" } },
],
},
},
],
},
{ choices: [{ delta: {}, finish_reason: "tool_calls" }] },
),
),
),
)
const stored = [
{ index: 0, type: "summary", summary: "Plan tools" },
{ index: 1, type: "encrypted", encrypted: "opaque" },
]
expect(response.reasoning).toBe("Let me think")
expect(response.message.content.find((part) => part.type === "reasoning")?.providerMetadata).toEqual({
openai: { reasoningField: "reasoning_content", reasoningDetails: stored },
})
const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] }))
expect(replay.body.messages).toEqual([
{
role: "assistant",
content: null,
tool_calls: [{ id: "call_1", type: "function", function: { name: "get_time", arguments: "{}" } }],
reasoning_content: "Let me think",
reasoning_details: [
{ type: "summary", summary: "Plan tools" },
{ type: "encrypted", encrypted: "opaque" },
],
},
])
}),
)
it.effect("displays Kimi summaries and replays reasoning_content when no scalar reasoning streams", () =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(request).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{ choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: "Plan" }] } }] },
{ choices: [{ delta: { reasoning_details: [{ index: 0, type: "summary", summary: " tools" }] } }] },
{ choices: [{ delta: { reasoning_details: [{ index: 1, type: "encrypted", encrypted: "opaque" }] } }] },
{ choices: [{ delta: { content: "Hello" } }] },
{ choices: [{ delta: {}, finish_reason: "stop" }] },
),
),
),
)
expect(response.reasoning).toBe("Plan tools")
const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] }))
expect(replay.body.messages).toEqual([
{
role: "assistant",
content: "Hello",
reasoning_content: "Plan tools",
reasoning_details: [
{ type: "summary", summary: "Plan tools" },
{ type: "encrypted", encrypted: "opaque" },
],
},
])
}),
)
// Sessions persisted before the fix already hold Kimi details with `index`.
it.effect("strips the streaming index from previously stored Kimi details", () =>
Effect.gen(function* () {
const replay = yield* compileRequest(
LLM.request({
model,
messages: [
Message.assistant([
{
type: "reasoning",
text: "thinking",
providerMetadata: {
openai: {
reasoningField: "reasoning_content",
reasoningDetails: [
{ index: 0, type: "summary", summary: "thinking" },
{ index: 1, type: "encrypted", encrypted: "opaque" },
],
},
},
},
]),
],
}),
)
expect(replay.body.messages).toEqual([
{
role: "assistant",
content: "",
reasoning_content: "thinking",
reasoning_details: [
{ type: "summary", summary: "thinking" },
{ type: "encrypted", encrypted: "opaque" },
],
},
])
}),
)
it.effect("merges consecutive OpenRouter summary deltas and replays them unmodified", () =>
Effect.gen(function* () {
const merged = [
{ type: "reasoning.summary", summary: "Plan tools", format: "openai-responses-v1", index: 0 },
{ type: "reasoning.encrypted", data: "opaque", format: "openai-responses-v1", index: 0 },
]
const response = yield* LLMClient.generate(request).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{
choices: [
{
delta: {
reasoning_details: [
{ type: "reasoning.summary", summary: "Plan", format: "openai-responses-v1", index: 0 },
],
},
},
],
},
{
choices: [
{
delta: {
reasoning_details: [
{ type: "reasoning.summary", summary: " tools", format: "openai-responses-v1", index: 0 },
],
},
},
],
},
{ choices: [{ delta: { reasoning_details: [merged[1]] } }] },
{ choices: [{ delta: { content: "Hello" } }] },
{ choices: [{ delta: {}, finish_reason: "stop" }] },
),
),
),
)
expect(response.reasoning).toBe("Plan tools")
expect(response.message.content.find((part) => part.type === "reasoning")?.providerMetadata).toEqual({
openai: { reasoningDetails: merged },
})
const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] }))
expect(replay.body.messages).toEqual([{ role: "assistant", content: "Hello", reasoning_details: merged }])
}),
)
it.effect("uses scalar display text for signature-only reasoning details", () =>
Effect.gen(function* () {
const details = [{ type: "reasoning.text", signature: "signed", format: "provider-v2", index: 0 }]
+10 -3
View File
@@ -315,10 +315,9 @@ describe("OpenRouter", () => {
}),
)
it.effect("preserves opaque and duplicate continuation details", () =>
it.effect("drops unrecognized details and preserves duplicate continuation details", () =>
Effect.gen(function* () {
const details = [
{ type: "reasoning.future", format: "provider-v2", state: { opaque: true } },
{ type: "reasoning.encrypted", id: "state", data: "opaque" },
{ type: "reasoning.encrypted", id: "state", data: "opaque" },
]
@@ -330,7 +329,15 @@ describe("OpenRouter", () => {
Message.assistant({
type: "reasoning",
text: "Thinking",
providerMetadata: { openrouter: { reasoningField: "reasoning", reasoningDetails: details } },
providerMetadata: {
openrouter: {
reasoningField: "reasoning",
reasoningDetails: [
{ type: "reasoning.future", format: "provider-v2", state: { opaque: true } },
...details,
],
},
},
}),
],
}),