From 057e06a9c36ef3a91a37661a072913b5c7e8a845 Mon Sep 17 00:00:00 2001 From: Yujun Liu Date: Thu, 24 Sep 2026 09:10:50 -0700 Subject: [PATCH] fix: replay Opus reasoning, stream thinking, keep inline system messages - Send signed prior thinking back to Copilot as reasoning_text and reasoning_opaque. Dropping it made Opus re-plan on every tool turn; the same Opus 5.5 xhigh Claude Code task went from 209-379s to ~100s. - Forward Copilot's reasoning_text/reasoning_opaque to Claude Code as thinking and signature deltas (and thinking blocks when not streaming), so long turns show progress instead of looking frozen. - Keep role:"system" messages inside messages[] (SessionStart hook context, reminders) as user context. They were stripped as an assistant prefill or attributed to the model. Co-Authored-By: Claude Opus 5.5 (1M context) --- src/lib/context-manager.ts | 2 + src/routes/messages/anthropic-types.ts | 17 +- src/routes/messages/non-stream-translation.ts | 96 +++++++++-- src/routes/messages/responses-bridge.ts | 16 +- src/routes/messages/stream-translation.ts | 82 +++++++--- .../copilot/create-chat-completions.ts | 10 ++ tests/anthropic-request.test.ts | 152 +++++++++++++----- tests/anthropic-response.test.ts | 140 ++++++++++++++++ tests/messages-responses-translation.test.ts | 20 +++ 9 files changed, 463 insertions(+), 72 deletions(-) diff --git a/src/lib/context-manager.ts b/src/lib/context-manager.ts index b22429b..9d62045 100644 --- a/src/lib/context-manager.ts +++ b/src/lib/context-manager.ts @@ -256,6 +256,8 @@ function estimateMessageBytes(msg: Message): number { } } if (msg.tool_call_id) bytes += msg.tool_call_id.length + 20 + if (msg.reasoning_text) bytes += msg.reasoning_text.length + 20 + if (msg.reasoning_opaque) bytes += msg.reasoning_opaque.length + 20 return bytes } diff --git a/src/routes/messages/anthropic-types.ts b/src/routes/messages/anthropic-types.ts index d4f633a..c31533b 100644 --- a/src/routes/messages/anthropic-types.ts +++ b/src/routes/messages/anthropic-types.ts @@ -21,6 +21,7 @@ export interface AnthropicMessagesPayload { thinking?: { type: "enabled" | "adaptive" budget_tokens?: number + display?: "summarized" | "omitted" } output_config?: { effort?: "low" | "medium" | "high" | "xhigh" | "max" @@ -82,6 +83,7 @@ export interface AnthropicToolUseBlock { export interface AnthropicThinkingBlock { type: "thinking" thinking: string + signature?: string } export type AnthropicUserContentBlock = @@ -105,7 +107,19 @@ export interface AnthropicAssistantMessage { content: string | Array } -export type AnthropicMessage = AnthropicUserMessage | AnthropicAssistantMessage +/** + * Mid-conversation system message. Claude Code sends these inside `messages` + * (e.g. SessionStart hook context), sometimes as the trailing entry. + */ +export interface AnthropicSystemMessage { + role: "system" + content: string | Array +} + +export type AnthropicMessage = + | AnthropicUserMessage + | AnthropicAssistantMessage + | AnthropicSystemMessage export interface AnthropicTool { name: string @@ -223,6 +237,7 @@ export interface AnthropicStreamState { messageStartSent: boolean contentBlockIndex: number contentBlockOpen: boolean + thinkingBlockOpen?: boolean toolCalls: { [openAIToolIndex: number]: { id: string diff --git a/src/routes/messages/non-stream-translation.ts b/src/routes/messages/non-stream-translation.ts index 554c8c8..35ca1da 100644 --- a/src/routes/messages/non-stream-translation.ts +++ b/src/routes/messages/non-stream-translation.ts @@ -18,7 +18,9 @@ import { type AnthropicMessage, type AnthropicMessagesPayload, type AnthropicResponse, + type AnthropicSystemMessage, type AnthropicTextBlock, + type AnthropicThinkingBlock, type AnthropicTool, type AnthropicToolResultBlock, type AnthropicToolUseBlock, @@ -346,15 +348,42 @@ function translateAnthropicMessagesToOpenAI( ): Array { const systemMessages = handleSystemPrompt(system) - const otherMessages = anthropicMessages.flatMap((message) => - message.role === "user" ? - handleUserMessage(message) - : handleAssistantMessage(message), - ) + const otherMessages = anthropicMessages.flatMap((message) => { + switch (message.role) { + case "user": { + return handleUserMessage(message) + } + case "system": { + return handleInlineSystemMessage(message) + } + default: { + return handleAssistantMessage(message) + } + } + }) return [...systemMessages, ...otherMessages] } +/** + * Claude Code sends hook context and reminders as `role: "system"` entries + * inside `messages`, often as the last one. Copilot chat completions has no + * mid-conversation system turn, and treating it as assistant either strips it + * as a prefill or attributes it to the model. Keep it in place as user + * context, which also leaves the cached system-prompt prefix untouched. + */ +function handleInlineSystemMessage( + message: AnthropicSystemMessage, +): Array { + const text = inlineSystemText(message) + return text ? [{ role: "user", content: text }] : [] +} + +export function inlineSystemText(message: AnthropicSystemMessage): string { + if (typeof message.content === "string") return message.content + return message.content.map((block) => block.text).join("\n\n") +} + // Reserved keywords that GitHub Copilot API blocks in system prompts // These will be completely removed from system prompts (case-insensitive) const RESERVED_KEYWORD_PATTERNS = [ @@ -443,17 +472,15 @@ function handleAssistantMessage( (block): block is AnthropicTextBlock => block.type === "text", ) - // Thinking blocks have signed-by-Anthropic semantics that Copilot can't - // verify or replay. Promoting their content to plain text would destroy - // the assistant turn semantics (the model would treat its own private - // reasoning as visible text). Drop them cleanly on the request side. const allTextContent = textBlocks.map((b) => b.text).join("\n\n") + const reasoning = getPriorReasoning(message.content) return toolUseBlocks.length > 0 ? [ { role: "assistant", content: allTextContent || null, + ...reasoning, tool_calls: toolUseBlocks.map((toolUse) => ({ id: toolUse.id, type: "function", @@ -468,10 +495,40 @@ function handleAssistantMessage( { role: "assistant", content: mapContent(message.content), + ...reasoning, }, ] } +/** + * Send the model's earlier reasoning back to Copilot the way it arrived: + * `reasoning_text` plus the `reasoning_opaque` signature Copilot issued with + * it (forwarded to Claude Code as the thinking block's signature). Dropping + * it makes Opus re-derive its plan from scratch on every tool-use turn: on + * captured Opus 5.5 xhigh turns that was 3-10x more output tokens. + * + * Only signed blocks are replayed, so reasoning that didn't come from Copilot + * (or a stripped block) is never presented to the model as its own. + */ +function getPriorReasoning( + content: Array, +): Pick { + const signed = content.filter( + (block): block is AnthropicThinkingBlock => + block.type === "thinking" && Boolean(block.signature), + ) + if (signed.length === 0) return {} + + const text = signed + .map((block) => block.thinking) + .filter(Boolean) + .join("\n\n") + return { + ...(text && { reasoning_text: text }), + reasoning_opaque: signed[0].signature, + } +} + function mapContent( content: | string @@ -500,7 +557,7 @@ function mapContent( break } - // Thinking blocks dropped — see handleAssistantMessage rationale. + // Thinking blocks travel as reasoning_text/reasoning_opaque, not content. case "image": { contentParts.push({ type: "image_url", @@ -596,10 +653,12 @@ export function translateToAnthropic( stopReason = response.choices[0]?.finish_reason ?? stopReason // Process all choices to extract text and tool use blocks + const allThinkingBlocks: Array = [] for (const choice of response.choices) { const textBlocks = getAnthropicTextBlocks(choice.message.content) const toolUseBlocks = getAnthropicToolUseBlocks(choice.message.tool_calls) + allThinkingBlocks.push(...getAnthropicThinkingBlocks(choice.message)) allTextBlocks.push(...textBlocks) allToolUseBlocks.push(...toolUseBlocks) @@ -609,14 +668,12 @@ export function translateToAnthropic( } } - // Note: GitHub Copilot doesn't generate thinking blocks, so we don't include them in responses - return { id: response.id, type: "message", role: "assistant", model: clientModel ?? response.model, - content: [...allTextBlocks, ...allToolUseBlocks], + content: [...allThinkingBlocks, ...allTextBlocks, ...allToolUseBlocks], stop_reason: mapOpenAIStopReasonToAnthropic(stopReason), stop_sequence: null, usage: { @@ -633,6 +690,19 @@ export function translateToAnthropic( } } +function getAnthropicThinkingBlocks( + message: ChatCompletionResponse["choices"][number]["message"], +): Array { + if (!message.reasoning_text && !message.reasoning_opaque) return [] + return [ + { + type: "thinking", + thinking: message.reasoning_text ?? "", + ...(message.reasoning_opaque && { signature: message.reasoning_opaque }), + }, + ] +} + function getAnthropicTextBlocks( messageContent: Message["content"], ): Array { diff --git a/src/routes/messages/responses-bridge.ts b/src/routes/messages/responses-bridge.ts index 3c11adc..f20770f 100644 --- a/src/routes/messages/responses-bridge.ts +++ b/src/routes/messages/responses-bridge.ts @@ -17,6 +17,7 @@ import type { AnthropicUserContentBlock, } from "./anthropic-types" +import { inlineSystemText } from "./non-stream-translation" import { translateReasoningEffort } from "./reasoning-effort" interface SSEStream { @@ -144,8 +145,19 @@ function translateSystem( function translateMessage( message: AnthropicMessagesPayload["messages"][number], ): Array { - if (message.role === "user") return translateUserMessage(message.content) - return translateAssistantMessage(message.content) + switch (message.role) { + case "user": { + return translateUserMessage(message.content) + } + case "system": { + // Mid-conversation system context (see handleInlineSystemMessage). + const text = inlineSystemText(message) + return text ? [{ role: "user", content: text }] : [] + } + default: { + return translateAssistantMessage(message.content) + } + } } function translateUserMessage( diff --git a/src/routes/messages/stream-translation.ts b/src/routes/messages/stream-translation.ts index 395498d..3f8a11c 100644 --- a/src/routes/messages/stream-translation.ts +++ b/src/routes/messages/stream-translation.ts @@ -16,6 +16,62 @@ function isToolBlockOpen(state: AnthropicStreamState): boolean { ) } +function closeOpenBlock( + state: AnthropicStreamState, + events: Array, +): void { + if (!state.contentBlockOpen) return + events.push({ type: "content_block_stop", index: state.contentBlockIndex }) + state.contentBlockIndex++ + state.contentBlockOpen = false + state.thinkingBlockOpen = false +} + +function openThinkingBlock( + state: AnthropicStreamState, + events: Array, +): void { + if (state.thinkingBlockOpen) return + closeOpenBlock(state, events) + events.push({ + type: "content_block_start", + index: state.contentBlockIndex, + content_block: { type: "thinking", thinking: "" }, + }) + state.contentBlockOpen = true + state.thinkingBlockOpen = true +} + +/** + * Copilot streams Claude's reasoning as `reasoning_text` deltas, then a single + * `reasoning_opaque` signature. Forward them as an Anthropic thinking block: + * on long high-effort turns this is the only output for minutes, and without + * it Claude Code sees a silent stream (pings don't count as progress). + */ +function translateReasoningDelta( + delta: ChatCompletionChunk["choices"][number]["delta"], + state: AnthropicStreamState, + events: Array, +): void { + if (delta.reasoning_text) { + openThinkingBlock(state, events) + events.push({ + type: "content_block_delta", + index: state.contentBlockIndex, + delta: { type: "thinking_delta", thinking: delta.reasoning_text }, + }) + } + + if (delta.reasoning_opaque) { + openThinkingBlock(state, events) + events.push({ + type: "content_block_delta", + index: state.contentBlockIndex, + delta: { type: "signature_delta", signature: delta.reasoning_opaque }, + }) + } +} + // eslint-disable-next-line max-lines-per-function, complexity export function translateChunkToAnthropicEvents( chunk: ChatCompletionChunk, @@ -57,15 +113,12 @@ export function translateChunkToAnthropicEvents( state.messageStartSent = true } + translateReasoningDelta(delta, state, events) + if (delta.content) { - if (isToolBlockOpen(state)) { - // A tool block was open, so close it before starting a text block. - events.push({ - type: "content_block_stop", - index: state.contentBlockIndex, - }) - state.contentBlockIndex++ - state.contentBlockOpen = false + if (isToolBlockOpen(state) || state.thinkingBlockOpen) { + // A tool or thinking block was open, so close it before starting text. + closeOpenBlock(state, events) } if (!state.contentBlockOpen) { @@ -93,16 +146,8 @@ export function translateChunkToAnthropicEvents( if (delta.tool_calls) { for (const toolCall of delta.tool_calls) { if (toolCall.id && toolCall.function?.name) { - // New tool call starting. - if (state.contentBlockOpen) { - // Close any previously open block. - events.push({ - type: "content_block_stop", - index: state.contentBlockIndex, - }) - state.contentBlockIndex++ - state.contentBlockOpen = false - } + // New tool call starting. Close any previously open block. + closeOpenBlock(state, events) const anthropicBlockIndex = state.contentBlockIndex state.toolCalls[toolCall.index] = { @@ -149,6 +194,7 @@ export function translateChunkToAnthropicEvents( index: state.contentBlockIndex, }) state.contentBlockOpen = false + state.thinkingBlockOpen = false } events.push( diff --git a/src/services/copilot/create-chat-completions.ts b/src/services/copilot/create-chat-completions.ts index d78dcec..be1ab02 100644 --- a/src/services/copilot/create-chat-completions.ts +++ b/src/services/copilot/create-chat-completions.ts @@ -140,6 +140,10 @@ export interface ChatCompletionChunk { interface Delta { content?: string | null role?: "user" | "assistant" | "system" | "tool" + /** Copilot streams Claude's extended thinking here while it reasons. */ + reasoning_text?: string | null + /** Signature for the reasoning, sent once when the thinking block ends. */ + reasoning_opaque?: string | null tool_calls?: Array<{ index: number id?: string @@ -180,6 +184,8 @@ export interface ChatCompletionResponse { interface ResponseMessage { role: "assistant" content: string | null + reasoning_text?: string | null + reasoning_opaque?: string | null tool_calls?: Array } @@ -267,6 +273,10 @@ export interface Message { name?: string tool_calls?: Array tool_call_id?: string + /** Prior Claude reasoning, sent back so the model can continue from it. */ + reasoning_text?: string + /** Signature Copilot issued with that reasoning (`reasoning_opaque`). */ + reasoning_opaque?: string } export interface ToolCall { diff --git a/tests/anthropic-request.test.ts b/tests/anthropic-request.test.ts index 51d2824..b701c28 100644 --- a/tests/anthropic-request.test.ts +++ b/tests/anthropic-request.test.ts @@ -126,46 +126,71 @@ describe("Anthropic to OpenAI translation logic", () => { expect(isValidChatCompletionRequest(openAIPayload)).toBe(false) }) - test("should drop thinking blocks from assistant messages (preserve text)", () => { - // Thinking blocks have signed-by-Anthropic semantics that Copilot can't - // verify. The proxy drops them on the request side rather than promoting - // their content to plain text (which would corrupt assistant turn - // semantics by exposing the model's private reasoning as visible output). - const anthropicPayload: AnthropicMessagesPayload = { - model: "claude-3-5-sonnet-20241022", + test("keeps a trailing inline system message as user context", () => { + // Claude Code sends SessionStart hook context as a trailing + // `role: "system"` message. It used to be mistaken for an assistant + // prefill and stripped, so the model never saw it. + const openAIPayload = translateToOpenAI({ + model: "claude-opus-5.5", + system: "You are Claude Code.", messages: [ - { role: "user", content: "What is 2+2?" }, + { role: "user", content: "Fix the bug." }, + { role: "system", content: "SessionStart hook additional context" }, + ], + max_tokens: 100, + }) + + expect(isValidChatCompletionRequest(openAIPayload)).toBe(true) + expect(openAIPayload.messages).toEqual([ + { role: "system", content: "You are Claude Code." }, + { role: "user", content: "Fix the bug." }, + { role: "user", content: "SessionStart hook additional context" }, + ]) + }) + + test("does not attribute a mid-conversation system message to the assistant", () => { + const openAIPayload = translateToOpenAI({ + model: "claude-opus-5.5", + messages: [ + { role: "user", content: "List files." }, + { + role: "system", + content: [{ type: "text", text: "hook context" }], + }, { role: "assistant", content: [ - { - type: "thinking", - thinking: "Let me think about this simple math problem...", - }, - { type: "text", text: "2+2 equals 4." }, + { type: "tool_use", id: "toolu_1", name: "Bash", input: {} }, + ], + }, + { + role: "user", + content: [ + { type: "tool_result", tool_use_id: "toolu_1", content: "a.ts" }, ], }, - { role: "user", content: "Thanks." }, ], max_tokens: 100, - } - const openAIPayload = translateToOpenAI(anthropicPayload) - expect(isValidChatCompletionRequest(openAIPayload)).toBe(true) + }) - const assistantMessage = openAIPayload.messages.find( - (m) => m.role === "assistant", - ) - // Text block preserved. - expect(assistantMessage?.content).toContain("2+2 equals 4.") - // Thinking block dropped — must not leak into outgoing payload. - expect(JSON.stringify(assistantMessage)).not.toContain( - "Let me think about this simple math problem", - ) + expect(openAIPayload.messages.map((m) => m.role)).toEqual([ + "user", + "user", + "assistant", + "tool", + ]) + expect(openAIPayload.messages[1].content).toBe("hook context") + expect(openAIPayload.messages[2].content).toBeNull() }) +}) - test("should drop thinking blocks but preserve text and tool calls", () => { +describe("Prior reasoning round-trip", () => { + test("sends signed prior thinking back as reasoning_text/reasoning_opaque", () => { + // Opus re-plans from scratch when its earlier reasoning is missing: on + // captured Opus 5.5 xhigh tool turns, dropping it cost 3-10x the output + // tokens. Copilot issued the signature, so it round-trips as-is. const anthropicPayload: AnthropicMessagesPayload = { - model: "claude-3-5-sonnet-20241022", + model: "claude-opus-5.5", messages: [ { role: "user", content: "What's the weather?" }, { @@ -173,8 +198,8 @@ describe("Anthropic to OpenAI translation logic", () => { content: [ { type: "thinking", - thinking: - "I need to call the weather API to get current weather information.", + thinking: "I need to call the weather API.", + signature: "sig-from-copilot", }, { type: "text", text: "I'll check the weather for you." }, { @@ -204,16 +229,67 @@ describe("Anthropic to OpenAI translation logic", () => { const assistantMessage = openAIPayload.messages.find( (m) => m.role === "assistant", ) - // Thinking block dropped. - expect(JSON.stringify(assistantMessage)).not.toContain( - "I need to call the weather API", + expect(assistantMessage).toMatchObject({ + content: "I'll check the weather for you.", + reasoning_text: "I need to call the weather API.", + reasoning_opaque: "sig-from-copilot", + }) + expect(assistantMessage?.tool_calls?.[0].function.name).toBe("get_weather") + }) + + test("never replays unsigned thinking or puts it in visible content", () => { + const anthropicPayload: AnthropicMessagesPayload = { + model: "claude-opus-5.5", + messages: [ + { role: "user", content: "What is 2+2?" }, + { + role: "assistant", + content: [ + { + type: "thinking", + thinking: "Let me think about this simple math problem...", + }, + { type: "text", text: "2+2 equals 4." }, + ], + }, + { role: "user", content: "Thanks." }, + ], + max_tokens: 100, + } + const openAIPayload = translateToOpenAI(anthropicPayload) + expect(isValidChatCompletionRequest(openAIPayload)).toBe(true) + + const assistantMessage = openAIPayload.messages.find( + (m) => m.role === "assistant", ) - // Text and tool calls preserved. - expect(assistantMessage?.content).toContain( - "I'll check the weather for you.", + expect(assistantMessage?.content).toBe("2+2 equals 4.") + expect(JSON.stringify(assistantMessage)).not.toContain( + "Let me think about this simple math problem", ) - expect(assistantMessage?.tool_calls).toHaveLength(1) - expect(assistantMessage?.tool_calls?.[0].function.name).toBe("get_weather") + }) + + test("keeps the signature when Claude Code omitted the thinking text", () => { + // Claude Code asks for display: "omitted"; the block can arrive with an + // empty thinking string but a valid signature. + const openAIPayload = translateToOpenAI({ + model: "claude-opus-5.5", + messages: [ + { role: "user", content: "hi" }, + { + role: "assistant", + content: [ + { type: "thinking", thinking: "", signature: "sig-only" }, + { type: "text", text: "hello" }, + ], + }, + { role: "user", content: "again" }, + ], + max_tokens: 100, + }) + + const assistantMessage = openAIPayload.messages[1] + expect(assistantMessage.reasoning_opaque).toBe("sig-only") + expect(assistantMessage).not.toHaveProperty("reasoning_text") }) }) diff --git a/tests/anthropic-response.test.ts b/tests/anthropic-response.test.ts index ecd71aa..3677b82 100644 --- a/tests/anthropic-response.test.ts +++ b/tests/anthropic-response.test.ts @@ -68,6 +68,32 @@ function isValidAnthropicStreamEvent(payload: unknown): boolean { } describe("OpenAI to Anthropic Non-Streaming Response Translation", () => { + test("puts Copilot reasoning in a leading thinking block", () => { + const response: ChatCompletionResponse = { + id: "msg_3", + object: "chat.completion", + created: 1, + model: "claude-opus-5.5", + choices: [ + { + index: 0, + message: { + role: "assistant", + content: "391", + reasoning_text: "17 * 23 = 391", + reasoning_opaque: "sig==", + }, + finish_reason: "stop", + logprobs: null, + }, + ], + } + + expect(translateToAnthropic(response).content).toEqual([ + { type: "thinking", thinking: "17 * 23 = 391", signature: "sig==" }, + { type: "text", text: "391" }, + ]) + }) test("should translate a simple text response correctly", () => { const openAIResponse: ChatCompletionResponse = { id: "chatcmpl-123", @@ -363,3 +389,117 @@ describe("OpenAI to Anthropic Streaming Response Translation", () => { } }) }) + +type ChunkDelta = ChatCompletionChunk["choices"][number]["delta"] +type ChunkFinish = ChatCompletionChunk["choices"][number]["finish_reason"] + +function opusChunk( + delta: ChunkDelta, + finishReason: ChunkFinish = null, +): ChatCompletionChunk { + return { + id: "msg_1", + object: "chat.completion.chunk", + created: 1, + model: "claude-opus-5.5", + choices: [{ index: 0, delta, finish_reason: finishReason, logprobs: null }], + } +} + +function newStreamState(): AnthropicStreamState { + return { + messageStartSent: false, + contentBlockIndex: 0, + contentBlockOpen: false, + toolCalls: {}, + } +} + +describe("Copilot reasoning stream translation", () => { + test("forwards reasoning as a thinking block before the tool call", () => { + // Opus 5.5 can reason for minutes before a tiny tool call. Copilot + // streams that reasoning as reasoning_text + reasoning_opaque; dropping it + // leaves Claude Code with a silent stream the whole time. + const streamState = newStreamState() + const events = [ + opusChunk({ content: null, reasoning_text: "Need to " }), + opusChunk({ content: null, reasoning_text: "list files" }), + opusChunk({ content: "", reasoning_opaque: "sig==" }), + opusChunk({ + tool_calls: [ + { + index: 0, + id: "toolu_1", + type: "function", + function: { name: "Bash", arguments: '{"command":"ls"}' }, + }, + ], + }), + opusChunk({}, "tool_calls"), + ].flatMap((c) => translateChunkToAnthropicEvents(c, streamState)) + + expect(events.map((e) => e.type)).toEqual([ + "message_start", + "content_block_start", + "content_block_delta", + "content_block_delta", + "content_block_delta", + "content_block_stop", + "content_block_start", + "content_block_delta", + "content_block_stop", + "message_delta", + "message_stop", + ]) + expect(events[1]).toEqual({ + type: "content_block_start", + index: 0, + content_block: { type: "thinking", thinking: "" }, + }) + expect(events[2]).toMatchObject({ + index: 0, + delta: { type: "thinking_delta", thinking: "Need to " }, + }) + expect(events[4]).toMatchObject({ + index: 0, + delta: { type: "signature_delta", signature: "sig==" }, + }) + expect(events[6]).toMatchObject({ + index: 1, + content_block: { type: "tool_use", id: "toolu_1", name: "Bash" }, + }) + expect(events[7]).toMatchObject({ index: 1 }) + }) + + test("closes the thinking block before visible text", () => { + const streamState = newStreamState() + const events = [ + opusChunk({ reasoning_text: "hmm" }), + opusChunk({ content: "No." }), + ].flatMap((c) => translateChunkToAnthropicEvents(c, streamState)) + + expect(events.slice(1)).toEqual([ + { + type: "content_block_start", + index: 0, + content_block: { type: "thinking", thinking: "" }, + }, + { + type: "content_block_delta", + index: 0, + delta: { type: "thinking_delta", thinking: "hmm" }, + }, + { type: "content_block_stop", index: 0 }, + { + type: "content_block_start", + index: 1, + content_block: { type: "text", text: "" }, + }, + { + type: "content_block_delta", + index: 1, + delta: { type: "text_delta", text: "No." }, + }, + ]) + }) +}) diff --git a/tests/messages-responses-translation.test.ts b/tests/messages-responses-translation.test.ts index 2738b5f..5d73314 100644 --- a/tests/messages-responses-translation.test.ts +++ b/tests/messages-responses-translation.test.ts @@ -105,6 +105,26 @@ describe("Messages Responses reasoning effort", () => { }) }) +describe("Messages Responses inline system messages", () => { + test("keeps inline system messages as user input, not assistant output", () => { + const result = translateAnthropicMessagesToResponses( + { + ...request, + messages: [ + { role: "user", content: "Explain the result" }, + { role: "system", content: "SessionStart hook additional context" }, + ], + }, + "gpt-5.5", + ) + + expect(result.input).toEqual([ + { role: "user", content: "Explain the result" }, + { role: "user", content: "SessionStart hook additional context" }, + ]) + }) +}) + interface ResponseCase { label: string response: ResponsesApiResponse