diff --git a/packages/app-acp/adapter-tool-calls-permissions.test.ts b/packages/app-acp/adapter-tool-calls-permissions.test.ts index f5cb1cdd..77cc496c 100644 --- a/packages/app-acp/adapter-tool-calls-permissions.test.ts +++ b/packages/app-acp/adapter-tool-calls-permissions.test.ts @@ -334,11 +334,11 @@ test("interrupted permission recovers on load with no effects or repeated reques expect(completions).toBe(1); expect(ran).toBe(0); expect(loaded.messages.some((m) => m.method === "session/request_permission")).toBe(false); - expect( - loaded - .updates() - .some((m) => m.update.sessionUpdate === "tool_call_update" && m.update.status === "failed"), - ).toBe(true); + // The journal holds the call and no outcome for it, so the card is drawn with no result. + expect(loaded.updates().map(({ update }) => update.sessionUpdate)).toContain("tool_call"); + expect(loaded.updates().some(({ update }) => update.sessionUpdate === "tool_call_update")).toBe( + false, + ); await loaded.close(); }); diff --git a/packages/app-acp/rpc/open.ts b/packages/app-acp/rpc/open.ts index ff6af1a5..a3cdcddd 100644 --- a/packages/app-acp/rpc/open.ts +++ b/packages/app-acp/rpc/open.ts @@ -3,7 +3,6 @@ import { isAbsolute } from "node:path"; import { RequestError, type AgentContext, type NewSessionRequest } from "@agentclientprotocol/sdk"; import { createSession, - projectConversation, restoreSession, SessionNotFoundError, type SessionOptions, @@ -28,7 +27,7 @@ import type { AdapterCore } from "./core.ts"; import { forwardPermission } from "./permission.ts"; import type { Session } from "./session.ts"; import type { SessionRegistry } from "./sessions.ts"; -import { toolEvidence, type SessionUpdates } from "./updates.ts"; +import { transcriptUpdates, type SessionUpdates } from "./updates.ts"; /** Everything opening a session needs from the connection. */ export type OpenDeps = Readonly<{ @@ -377,13 +376,8 @@ export function opener(deps: OpenDeps): OpenSession { sessions.set(sessionId, commandEntry); entry.revision = runtime.snapshot.durable.revision; if (id && replay) { - const durable = runtime.snapshot.durable; - const view = projectConversation(durable); - const evidence = toolEvidence(durable); - updates.replay(client, id, view.context, `${id}/context`, evidence, renderers); - view.log.forEach((turn, index) => { - updates.replay(client, id, turn.messages, `${id}/history/${index}`, evidence, renderers); - }); + for (const update of transcriptUpdates(runtime.snapshot.durable, renderers)) + core.send(client, id, update); } const registry = runtime.registry; if (registry.kind === "pending_adoption") diff --git a/packages/app-acp/rpc/updates.ts b/packages/app-acp/rpc/updates.ts index 3decf642..3cd3111f 100644 --- a/packages/app-acp/rpc/updates.ts +++ b/packages/app-acp/rpc/updates.ts @@ -1,13 +1,23 @@ -import type { AgentContext, SessionUpdate } from "@agentclientprotocol/sdk"; -import { blobUri } from "@labkit/core-agent"; +import type { AgentContext, ContentBlock, SessionUpdate } from "@agentclientprotocol/sdk"; +import { blobUri, projectBlob, projectMessage } from "@labkit/core-agent"; import type { + JournalBody, JournalState, MessageView, SessionBindings, SessionRuntime, SessionState, + ViewBlob, + WireEvent, } from "@labkit/core-agent"; -import type { HostToolNotification } from "@labkit/core-agent/host"; +import type { BlobRef } from "@labkit/core-agent/content"; +import { + settledTool, + toolAnnouncement, + toolCallIdOf, + type HostToolNotification, + type ToolIdentity, +} from "@labkit/core-agent/host"; import { diagnostic, diagnosticError } from "@labkit/core-agent/logging"; import type { AcpOptions } from "../adapter.ts"; @@ -28,14 +38,6 @@ export type SessionUpdates = Readonly<{ snapshot: SessionState, ): void; refreshInfo(entry: Session, client: AgentContext): void; - replay( - client: AgentContext, - id: string, - messages: readonly MessageView[], - prefix: string, - evidence: Map, - renderers: ReadonlyMap, - ): void; bindings( entry: Omit & { runtime?: SessionRuntime }, client: AgentContext, @@ -64,6 +66,7 @@ function toolUpdate( event: HostToolNotification, terminals: readonly string[] = [], renderers: ReadonlyMap = new Map(), + reconstructed = false, ): SessionUpdate { if (event.sessionUpdate === "tool_call") return { @@ -83,6 +86,7 @@ function toolUpdate( ...(event.rawOutput !== undefined ? { rawOutput: event.rawOutput, + ...(reconstructed ? { _meta: { "labkit.dev/reconstructed": true } } : {}), content: [ ...terminals.map((terminalId) => ({ type: "terminal" as const, terminalId })), ...renderToolContent( @@ -95,7 +99,7 @@ function toolUpdate( turnId: event.turnId, batchId: event.batchId, callId: event.callId, - reconstructed: false, + reconstructed, }, ), ...(event.name && renderers.has(event.name) @@ -122,6 +126,40 @@ function toolUpdate( }; } +/** + * The `session/update` for one tool notification, as a client is sent it: the card keeps the title + * the call was announced with, located once locations arrive, and its last status. Forgets the card + * once the call settles. + */ +function showTool( + cards: Session["toolCards"], + terminals: Session["terminals"], + event: HostToolNotification, + renderers: ReadonlyMap, + reconstructed = false, +): SessionUpdate { + if (event.sessionUpdate === "tool_call") + cards.set(event.toolCallId, { + baseTitle: event.title, + title: event.title, + status: event.status, + }); + const card = cards.get(event.toolCallId); + if (card && event.sessionUpdate === "tool_call_update") { + if (event.locations) card.title = locatedTitle(card.baseTitle, event.locations); + if (event.status) card.status = event.status; + } + const update = { + ...toolUpdate(event, terminals.get(event.toolCallId), renderers, reconstructed), + ...(card ? { title: card.title, status: card.status } : {}), + }; + if (event.status === "completed" || event.status === "failed") { + terminals.delete(event.toolCallId); + cards.delete(event.toolCallId); + } + return update; +} + function observeSafely( callback: ((value: T) => unknown) | undefined, value: T, @@ -139,51 +177,190 @@ function observeSafely( } } -/** Raw journal tool outcomes retain failures even when policy projects them as tool text. */ -export function toolEvidence(state: JournalState) { - const evidence = new Map(); - const owners = new Map(); +type ToolRecord = Extract; +type Settled = Extract["event"], { type: "model_settled" }>; + +/** + * The `session/update`s that draw a saved session, read from its journal records in order: each + * prompt; each step's recorded thinking, answer text and tool calls; each call's recorded outcome + * and how its card was shown; and what the stream showed of a step that did not finish. Messages a + * fork or compaction started from are in the `created` record and are drawn first. `renderers` + * format a tool's recorded output as it was formatted live. + */ +export function transcriptUpdates( + state: JournalState, + renderers: ReadonlyMap, +): SessionUpdate[] { + const out: SessionUpdate[] = []; + const cards: Session["toolCards"] = new Map(); + const terminals: Session["terminals"] = new Map(); + const sessionId = state.conversation.sessionId; + const chunk = ( + sessionUpdate: "user_message_chunk" | "agent_message_chunk" | "agent_thought_chunk", + content: ContentBlock, + messageId?: string, + ) => out.push({ sessionUpdate, ...(messageId ? { messageId } : {}), content } as SessionUpdate); + const card = (event: HostToolNotification) => + out.push(showTool(cards, terminals, event, renderers, true)); + const announced = new Map(); + const created = state.records[0]?.body; - let historyIndex = created?.kind === "created" ? created.seed.log.length : 0; - let active: { owner: string; turnId: string } | undefined; - for (const { body } of state.records) { - if (body.kind === "event" && body.event.type === "child") { + if (created?.kind === "created") { + const seeded = [...created.seed.context, ...created.seed.log.flatMap((turn) => turn.messages)]; + seededUpdates(out, sessionId, seeded.map(projectMessage), renderers); + } + + const bodies = state.records.map(({ body }) => body); + const toolOf = new Map(); + for (const body of bodies) + if (body.kind === "tool") toolOf.set(`${body.turnId} ${body.batchId} ${body.callId}`, body); + // The batch a step's calls ran in: the first batch of the same turn recorded after the step. + const batchAfter = (from: number, turnId: string): string | undefined => { + for (const body of bodies.slice(from + 1)) { + if (body.kind === "tool" && body.turnId === turnId) return body.batchId; + if (body.kind !== "event" || body.event.type !== "child" || body.event.turnId !== turnId) + continue; const event = body.event.event; - if (event.type === "model_settled" && event.result.kind === "succeeded") { - const previous = owners.get(body.event.turnId) ?? []; - previous.push(event.child.id); - owners.set(body.event.turnId, previous); - if (event.result.value.kind === "tools") - active = { owner: event.child.id, turnId: body.event.turnId }; - } + if (event.type === "model_settled") return undefined; + if (event.type === "batch_settled") return event.child.id; + if (event.type === "failed" && event.child.kind === "batch") return event.child.id; + } + return undefined; + }; + + for (const [index, body] of bodies.entries()) { + if (body.kind === "queued") prompt(body.text, body.attachments); + if (body.kind === "event" && body.event.type === "user") + prompt(body.event.text, body.event.attachments); + if ( + body.kind === "event" && + body.event.type === "child" && + body.event.event.type === "model_settled" + ) + step(index, body.event.turnId, body.event.event); + if (body.kind === "tool") { + const identity = announced.get(toolCallIdOf(body.batchId, body.callId)); + if (identity) card(settledTool(identity, body.result)); } - if (body.kind === "tool" && active?.turnId === body.turnId) - evidence.set( - `${active.owner}/${body.callId}`, - body.result.kind === "succeeded" ? "completed" : "failed", + } + return out; + + function prompt(text: string, attachments: readonly BlobRef[] | undefined) { + // A prompt is identified by its place in the transcript, as the client numbers the one it sent. + if (text) chunk("user_message_chunk", { type: "text", text }); + for (const blob of attachments ?? []) + chunk("user_message_chunk", resourceLink(projectBlob(blob))); + } + + function step(index: number, turnId: ToolIdentity["turnId"], event: Settled) { + const completionId = event.child.id; + if (event.shown?.thinking) + chunk( + "agent_thought_chunk", + { type: "text", text: event.shown.thinking }, + `${completionId}/thought`, ); - if (body.kind === "terminal") { - const completed = owners.get(body.turnId) ?? []; - let assistant = 0; - // The fold appends exactly one log entry per terminal record, in the same order this loop - // sees them, so `historyIndex` names the entry this record produced. Terminal records carry - // only turnId, agent and outcome; the turn's messages come from the fold's log entry. - state.conversation.log[historyIndex]?.messages.forEach((message, index) => { - if (message.role !== "assistant") return; - const owner = completed[assistant++]; - for (const call of message.calls ?? []) { - const status = evidence.get(`${owner}/${call.id}`); - if (status) - evidence.set( - `${state.conversation.sessionId}/history/${historyIndex}/${index}/${call.id}`, - status, - ); - } + if (event.result.kind !== "succeeded") { + if (event.shown?.text) + chunk("agent_message_chunk", { type: "text", text: event.shown.text }, completionId); + return; + } + const completion = event.result.value; + if (completion.text) + chunk("agent_message_chunk", { type: "text", text: completion.text }, completionId); + if (completion.kind !== "tools") return; + const batchId = batchAfter(index, turnId); + for (const call of completion.calls) { + const record = batchId ? toolOf.get(`${turnId} ${batchId} ${call.id}`) : undefined; + const identity: ToolIdentity = { + sessionId, + turnId, + batchId: (batchId ?? completionId) as ToolIdentity["batchId"], + callId: call.id, + // A call whose batch never started has no recorded ID; it is drawn under its step's. + toolCallId: batchId + ? toolCallIdOf(batchId, call.id) + : (`${completionId}/tool/${call.id}` as ToolIdentity["toolCallId"]), + name: call.name, + }; + announced.set(identity.toolCallId, identity); + card(toolAnnouncement(identity, call, record?.shown)); + if (record?.shown?.locations) + card({ ...identity, sessionUpdate: "tool_call_update", locations: record.shown.locations }); + } + } +} + +function resourceLink(blob: ViewBlob): ContentBlock { + return { + type: "resource_link", + uri: blob.uri, + name: blob.name ?? blob.id, + mimeType: blob.media, + size: blob.bytes, + }; +} + +/** + * Messages a fork or compaction started from, as its `created` record holds them: text, blobs and + * tool calls with the results the parent's history gave the model. Their cards carry IDs made from + * the message's place, since the parent's call IDs are not part of this record. + */ +function seededUpdates( + out: SessionUpdate[], + sessionId: string, + messages: readonly MessageView[], + renderers: ReadonlyMap, +) { + const calls = new Map(); + for (const [index, message] of messages.entries()) { + const messageId = `${sessionId}/context/${index}`; + if (message.role === "user" || message.role === "assistant") { + const sessionUpdate = message.role === "user" ? "user_message_chunk" : "agent_message_chunk"; + const named = message.role === "assistant" ? { messageId } : {}; + if (message.text) + out.push({ sessionUpdate, ...named, content: { type: "text", text: message.text } }); + for (const blob of message.blobs) + out.push({ sessionUpdate, ...named, content: resourceLink(blob) }); + } + if (message.role === "assistant") + for (const call of message.calls ?? []) { + const toolCallId = `${messageId}/tool/${call.id}`; + calls.set(call.id, { toolCallId, toolName: call.name }); + out.push({ + sessionUpdate: "tool_call", + toolCallId, + title: call.name, + name: call.name, + rawInput: call.args, + }); + } + if (message.role === "tool") { + const call = calls.get(message.callId); + if (!call) continue; + out.push({ + sessionUpdate: "tool_call_update", + toolCallId: call.toolCallId, + rawOutput: message.text, + _meta: { "labkit.dev/reconstructed": true }, + content: [ + ...renderToolContent(undefined, message.text, { + sessionId, + toolCallId: call.toolCallId, + toolName: call.toolName, + reconstructed: true, + }), + ...(renderers.has(call.toolName) + ? [] + : message.blobs.map((blob) => ({ + type: "content" as const, + content: resourceLink(blob), + }))), + ], }); - historyIndex++; + calls.delete(message.callId); } } - return evidence; } /** Builds the session/update projection for one connection. */ @@ -278,102 +455,9 @@ export function sessionUpdates( } }; - const replay = ( - client: AgentContext, - id: string, - messages: readonly MessageView[], - prefix: string, - evidence: Map, - renderers: ReadonlyMap, - ) => { - const calls = new Map< - string, - { toolCallId: string; toolName: string; status?: "completed" | "failed" } - >(); - messages.forEach((message, index) => { - const messageId = - message.role === "assistant" && message.owner - ? `${message.owner.turnId}/${message.owner.generation}` - : `${prefix}/${index}`; - if (message.role === "user" || message.role === "assistant") { - if (message.text) - core.send(client, id, { - sessionUpdate: message.role === "user" ? "user_message_chunk" : "agent_message_chunk", - messageId, - content: { type: "text", text: message.text }, - }); - for (const blob of message.blobs) - core.send(client, id, { - sessionUpdate: message.role === "user" ? "user_message_chunk" : "agent_message_chunk", - messageId, - content: { - type: "resource_link", - uri: blob.uri, - name: blob.name ?? blob.id, - mimeType: blob.media, - size: blob.bytes, - }, - }); - } - if (message.role === "assistant") - for (const call of message.calls ?? []) { - const toolCallId = `${messageId}/tool/${call.id}`; - calls.set(call.id, { - toolCallId, - toolName: call.name, - status: evidence.get(`${messageId}/${call.id}`), - }); - core.send(client, id, { - sessionUpdate: "tool_call", - toolCallId, - title: call.name, - name: call.name, - rawInput: call.args, - kind: "other", - }); - } - if (message.role === "tool") { - const call = calls.get(message.callId); - if (call) { - const { toolCallId, toolName, status } = call; - core.send(client, id, { - sessionUpdate: "tool_call_update", - toolCallId, - ...(status ? { status } : {}), - rawOutput: message.text, - _meta: { "labkit.dev/reconstructed": true }, - content: [ - ...renderToolContent( - status === "completed" ? renderers.get(toolName) : undefined, - message.text, - { sessionId: id, toolCallId, toolName, reconstructed: true }, - ), - ...(renderers.has(toolName) - ? [] - : message.blobs.map((blob) => ({ - type: "content" as const, - content: { - type: "resource_link" as const, - uri: blob.uri, - name: blob.name ?? blob.media, - mimeType: blob.media, - size: blob.bytes, - }, - }))), - ], - }); - calls.delete(message.callId); - } - } - }); - for (const { toolCallId } of calls.values()) - core.send(client, id, { sessionUpdate: "tool_call_update", toolCallId, status: "failed" }); - }; - return { refreshInfo, observe, - replay, bindings: (entry, client, renderers, subscribers, boundSessionId) => ({ observe: (snapshot) => { const bound = boundSessionId(); @@ -386,26 +470,8 @@ export function sessionUpdates( }); }, toolUpdate: (event) => { - if (event.sessionUpdate === "tool_call") - entry.toolCards.set(event.toolCallId, { - baseTitle: event.title, - title: event.title, - status: event.status, - }); - const card = entry.toolCards.get(event.toolCallId); - if (card && event.sessionUpdate === "tool_call_update") { - if (event.locations) card.title = locatedTitle(card.baseTitle, event.locations); - if (event.status) card.status = event.status; - } - if (entry.acceptingUpdates && event.sessionId) - core.send(client, event.sessionId, { - ...toolUpdate(event, entry.terminals.get(event.toolCallId), renderers), - ...(card ? { title: card.title, status: card.status } : {}), - }); - if (event.status === "completed" || event.status === "failed") { - entry.terminals.delete(event.toolCallId); - entry.toolCards.delete(event.toolCallId); - } + const update = showTool(entry.toolCards, entry.terminals, event, renderers); + if (entry.acceptingUpdates && event.sessionId) core.send(client, event.sessionId, update); observeSafely(subscribers.toolUpdate, event, { connectionId, sessionId: event.sessionId, diff --git a/packages/app-acp/view-model-parity.test.ts b/packages/app-acp/view-model-parity.test.ts index 338d94fd..a50ffe62 100644 --- a/packages/app-acp/view-model-parity.test.ts +++ b/packages/app-acp/view-model-parity.test.ts @@ -10,6 +10,8 @@ import { initialState, reduce, type ViewEvent } from "@labkit/view-model"; import { z } from "zod"; import { until } from "../core-agent/agent/test-support.ts"; +import { anthropicMessagesV3, openaiChatV2 } from "../core-agent/providers/index.ts"; +import { streamResponse, streamVector } from "../core-agent/providers/testing/stream-vectors.ts"; import { acpHttpHandler } from "./http.ts"; import { answer, configurable, tools } from "./testing/fixtures.ts"; import { setup } from "./testing/harness.ts"; @@ -18,42 +20,17 @@ const token = `${crypto.randomUUID()}${crypto.randomUUID()}`; const url = "http://acp.test/acp"; /** - * Whether a tool card's raw output says the call was refused. Live the output is an object and after - * `session/load` it is the saved result text, with different wording for the reason, so only the - * fact of refusal is comparable. + * What a reader sees: every block, and every tool card whole. `_meta` is left out: after + * `session/load` it carries `labkit.dev/reconstructed`, which marks a card as restored; nothing else + * may differ. */ -function refusedIn(rawOutput: unknown): boolean | undefined { - const value = typeof rawOutput === "string" ? safeParse(rawOutput) : rawOutput; - if (value === null || typeof value !== "object") return undefined; - return (value as { refused?: unknown }).refused === true; -} - -function safeParse(text: string): unknown { - try { - return JSON.parse(text); - } catch { - return undefined; - } -} - -/** What a reader sees: block kinds and text, and each tool card's title, status and whether it was refused. */ function seen(state: ReturnType) { - return state.blocks.map((block) => { - if (block.kind === "tool") { - const card = state.toolCalls[block.toolCallId]; - return { - kind: "tool", - title: card?.title, - status: card?.status, - refused: refusedIn(card?.rawOutput), - }; - } - if ("content" in block) { - const text = block.content.map((part) => (part.type === "text" ? part.text : part.type)); - return { kind: block.kind, text: text.join("") }; - } - return { kind: block.kind }; - }); + return { + blocks: state.blocks, + toolCalls: Object.fromEntries( + Object.entries(state.toolCalls).map(([id, { _meta, ...card }]) => [id, card]), + ), + }; } function host(options: Parameters[0]) { @@ -87,27 +64,46 @@ async function open(fetcher: typeof fetch, sessionId?: string) { return { client, state: () => state, events }; } -test("a plain answer looks the same live and after session/load", async () => { - const { options } = setup(); - const { handler, fetch } = host(options); - const live = await open(fetch); - await live.client.prompt("Go"); - const before = seen(live.state()); - expect(before).toEqual([ - { kind: "user", text: "Go" }, - { kind: "assistant", text: "Hello 🌍" }, - ]); - const id = live.client.sessionId; - await live.client.close(); +/** Follows one prompt live, answering every permission request with the first option of `kind`. */ +async function answering(fetcher: typeof fetch, kind: "allow_once" | "reject_once") { + let state = initialState; + const client = await connectSession({ + url, + fetch: fetcher, + onEvent: (event) => { + state = reduce(state, event); + if (event.type === "permission_requested") { + const option = event.request.options.find((candidate) => candidate.kind === kind); + void client.answerPermission(event.requestId, { + outcome: "selected", + optionId: option!.optionId, + }); + } + }, + }); + await client.prompt("Go"); + const id = client.sessionId; + await client.close(); + return { id, state }; +} - const reopened = await open(fetch, id); - await until(() => seen(reopened.state()).length >= before.length); - expect(seen(reopened.state())).toEqual(before); +async function reopen(fetcher: typeof fetch, id: string, blocks: number) { + const reopened = await open(fetcher, id); + await until(() => reopened.state().blocks.length >= blocks); await reopened.client.close(); + return reopened.state(); +} + +test("a plain answer is the same live and after session/load", async () => { + const { options } = setup(); + const { handler, fetch } = host(options); + const { id, state } = await answering(fetch, "allow_once"); + expect(state.blocks.map((block) => block.kind)).toEqual(["user", "assistant"]); + expect(seen(await reopen(fetch, id, state.blocks.length))).toEqual(seen(state)); await handler.close(); }); -test("a refused tool call and the reply after it look the same live and after session/load", async () => { +test("a refused tool call and the reply after it are the same live and after session/load", async () => { let completions = 0; const { options } = setup({ complete: () => (++completions === 1 ? tools : answer), @@ -116,42 +112,76 @@ test("a refused tool call and the reply after it look the same live and after se ]), }); const { handler, fetch } = host(options); + const { id, state } = await answering(fetch, "reject_once"); + expect(state.blocks.map((block) => block.kind)).toEqual([ + "user", + "assistant", + "tool", + "assistant", + ]); + expect(Object.values(state.toolCalls)).toMatchObject([ + { status: "failed", rawOutput: { refused: true } }, + ]); + expect(seen(await reopen(fetch, id, state.blocks.length))).toEqual(seen(state)); + await handler.close(); +}); - let respond: (() => void) | undefined; - let state = initialState; - const client = await connectSession({ - url, - fetch, - onEvent: (event) => { - state = reduce(state, event); - if (event.type === "permission_requested") { - const reject = event.request.options.find((option) => option.kind === "reject_once"); - respond = () => - client.answerPermission(event.requestId, { - outcome: "selected", - optionId: reject!.optionId, - }); - } - }, - }); - const turn = client.prompt("Go"); - await until(() => respond !== undefined); - respond!(); - await turn; - - const live = seen(state); - expect(live.map((item) => item.kind)).toEqual(["user", "assistant", "tool", "assistant"]); - expect(live.find((item) => item.kind === "tool")).toMatchObject({ - status: "failed", - refused: true, +test("a located read and a failed call are the same live and after session/load", async () => { + let completions = 0; + const base = setup({ + complete: () => + ++completions === 1 + ? { + kind: "tools", + text: "Looking", + calls: [ + { id: "one", name: "look", args: { path: "/workspace/notes.md" } }, + { id: "two", name: "boom", args: {} }, + ], + } + : answer, + tools: new Map([ + [ + "look", + defineTool({ + input: z.object({ path: z.string() }), + kind: "read", + locations: ({ path }) => [{ path, line: 3 }], + run: () => "contents", + }), + ], + [ + "boom", + defineTool({ + input: z.object({}), + run: () => { + throw new Error("disk on fire"); + }, + }), + ], + ]), }); - const id = client.sessionId; - await client.close(); - - const reopened = await open(fetch, id); - await until(() => seen(reopened.state()).length >= live.length); - expect(seen(reopened.state())).toEqual(live); - await reopened.client.close(); + const options: typeof base.options = { + ...base.options, + sessionOptions: async (context) => { + const original = await base.options.sessionOptions(context); + return { + ...original, + configuration: { + ...original.configuration, + agents: new Map([["a", { model: "m", tools: ["look", "boom"] }]]), + policy: { toolFailure: "return-error-and-continue" }, + }, + }; + }, + }; + const { handler, fetch } = host(options); + const { id, state } = await answering(fetch, "allow_once"); + expect(Object.values(state.toolCalls)).toMatchObject([ + { kind: "read", status: "completed", locations: [{ path: "/workspace/notes.md", line: 3 }] }, + { status: "failed", rawOutput: { error: "disk on fire" } }, + ]); + expect(seen(await reopen(fetch, id, state.blocks.length))).toEqual(seen(state)); await handler.close(); }); @@ -174,3 +204,51 @@ test("the options a session opens with are shown, a selection updates them, and await reopened.client.close(); await handler.close(); }); + +for (const profile of [openaiChatV2, anthropicMessagesV3]) { + test(`${profile.id}: streamed thinking and answer are the same live and after session/load`, async () => { + const base = setup(); + const options: typeof base.options = { + ...base.options, + sessionOptions: async (context) => { + const original = await base.options.sessionOptions(context); + return { + ...original, + configuration: { + ...original.configuration, + policy: { + provider: profile.id, + model: "m", + stream: true, + thinking: profile.capabilities.thinking.mode === "budget" ? "budget" : "high", + thinkingBudgetTokens: profile.capabilities.thinking.mode === "budget" ? 1024 : null, + maxOutputTokens: 4096, + }, + }, + bindings: { + ...original.bindings, + complete: undefined, + providers: new Map([ + [ + profile.id, + { + profile, + transport: { + baseUrl: "https://example.invalid", + fetch: (async () => + streamResponse(streamVector(profile))) as unknown as typeof fetch, + }, + }, + ], + ]), + }, + }; + }, + }; + const { handler, fetch } = host(options); + const { id, state } = await answering(fetch, "allow_once"); + expect(state.blocks.map((block) => block.kind)).toEqual(["user", "thought", "assistant"]); + expect(seen(await reopen(fetch, id, state.blocks.length))).toEqual(seen(state)); + await handler.close(); + }); +} diff --git a/packages/core-agent/agent/agent-fsm.ts b/packages/core-agent/agent/agent-fsm.ts index 40c99cf1..e7dc57ae 100644 --- a/packages/core-agent/agent/agent-fsm.ts +++ b/packages/core-agent/agent/agent-fsm.ts @@ -1,7 +1,7 @@ import type { z } from "zod"; import { defineMachine, stay, type Decision } from "../fsm/fsm.ts"; -import type { Continuation } from "../providers/types.ts"; +import type { Continuation, Shown } from "../providers/types.ts"; import type { CompletionUsage } from "../providers/usage.ts"; import { validatePermissionDecisions, type PermissionDecisions } from "./permissions.ts"; import type { BatchOutcome } from "./tool-batch.ts"; @@ -104,6 +104,7 @@ export type TurnEvent = * @property provider Provider that produced this output, when the policy named one. * @property model Application model name that produced this output. * @property continuation Provider continuation payload (thinking signatures), not the next step. + * @property shown What the client was shown while the step streamed; see `ShownSchema`. * @property permissionRequired Ask the user before running the proposed tool calls. */ | { @@ -113,6 +114,7 @@ export type TurnEvent = provider?: string; model?: string; continuation?: Continuation; + shown?: Shown; usage?: CompletionUsage; permissionRequired?: true; } diff --git a/packages/core-agent/agent/tool-batch.ts b/packages/core-agent/agent/tool-batch.ts index 6b48c2b2..b588b777 100644 --- a/packages/core-agent/agent/tool-batch.ts +++ b/packages/core-agent/agent/tool-batch.ts @@ -98,13 +98,16 @@ export type BatchCommand = | { type: "cancel_tool"; child: Ref<"tool">; reason?: Failure } | { type: "notify"; outcome: BatchOutcome }; +/** ID of the `tool` child that runs call `callId` of tool batch `batchId`. */ +export const toolOperationId = (batchId: string, callId: string) => `${batchId}/${callId}`; + /** * Builds the tool batch machine for batch `id`. Each call runs as a `tool` child with ref * `/`. The batch settles once: when every call succeeded, or at the first failed or * cancelled call. */ export function toolBatchMachine(id: Ref<"batch">) { - const toolRef = (callId: string) => ref("tool", `${id.id}/${callId}`); + const toolRef = (callId: string) => ref("tool", toolOperationId(id.id, callId)); const settle = ( pending: ToolCalls, outcome: BatchOutcome, diff --git a/packages/core-agent/host/commands/model.ts b/packages/core-agent/host/commands/model.ts index 8572f09e..35b03bfe 100644 --- a/packages/core-agent/host/commands/model.ts +++ b/packages/core-agent/host/commands/model.ts @@ -40,8 +40,14 @@ export function completeModel( }; let status = "pending"; let stream = false; + // What the client was sent of this step's stream, recorded on `model_settled` whatever the + // step's outcome, so a reopened session shows exactly what was shown live. + let shownThinking = ""; + let shownText = ""; const notifyStream = (fields: Omit) => { - if (!host.closed && stream) notify(host.streamUpdate, { ...identity, ...fields }); + if (host.closed || !stream) return false; + notify(host.streamUpdate, { ...identity, ...fields }); + return true; }; host.spawn( @@ -138,8 +144,15 @@ export function completeModel( blobs, (delta) => { const parsed = StreamDeltaSchema.safeParse(delta); - if (parsed.success && !signal.aborted && status === "in_progress") - notifyStream({ ...parsed.data, sessionUpdate: "completion_update" }); + if ( + parsed.success && + !signal.aborted && + status === "in_progress" && + notifyStream({ ...parsed.data, sessionUpdate: "completion_update" }) + ) { + shownThinking += parsed.data.thinking ?? ""; + shownText += parsed.data.text ?? ""; + } }, { sessionId: host.sessionId, @@ -203,6 +216,11 @@ export function completeModel( ...(result.kind === "succeeded" && result.value.continuation ? { continuation: result.value.continuation } : {}), + shown: { + ...(shownThinking ? { thinking: shownThinking } : {}), + // A settled answer's text is the completion itself; only an unfinished one needs its own. + ...(shownText && result.kind !== "succeeded" ? { text: shownText } : {}), + }, }), (state) => { const next = diff --git a/packages/core-agent/host/commands/permission.ts b/packages/core-agent/host/commands/permission.ts index ccb470e4..0cedf6ff 100644 --- a/packages/core-agent/host/commands/permission.ts +++ b/packages/core-agent/host/commands/permission.ts @@ -2,12 +2,13 @@ import { z } from "zod"; import type { TurnCommand } from "../../agent/agent-fsm.ts"; import { PermissionDecisionsSchema, type PermissionDecisions } from "../../agent/permissions.ts"; -import { failure, ref, type ActorId, type Failure } from "../../agent/types.ts"; +import { failure, type ActorId, type Failure } from "../../agent/types.ts"; import { freeze } from "../../fsm/fsm.ts"; import { diagnosticError } from "../../logging/index.ts"; import type { HostContext } from "../context.ts"; import type { ExecutionContext, HostToolNotification } from "../host.ts"; -import { PermissionResponseSchema, ToolLocationSchema, type ToolLocation } from "../ports.ts"; +import { PermissionResponseSchema, type ToolLocation } from "../ports.ts"; +import { toolAnnouncement, toolCallIdOf, toolLocations } from "../tool-display.ts"; /** * Asks the permission port about each tool call of a completion, in order, and records the @@ -29,6 +30,7 @@ export function requestPermission( refused: new Map(), pending: [] as HostToolNotification[], remembered: new Map(), + locations: new Map(), }; host.grants.set(command.child.id, grant); host.spawn( @@ -57,18 +59,10 @@ export function requestPermission( turnId, batchId: command.batch.id, callId: call.id, - toolCallId: ref("tool", `${command.batch.id}/${call.id}`).id, + toolCallId: toolCallIdOf(command.batch.id, call.id), name: call.name, }; - const display = { - ...identity, - sessionUpdate: "tool_call" as const, - title: call.name, - name: call.name, - kind: tool.kind ?? "other", - status: "pending" as const, - rawInput: call.args, - }; + const display = toolAnnouncement(identity, call, tool); grant.pending.push(display); host.notifyTool(display); signal.throwIfAborted(); @@ -78,9 +72,7 @@ export function requestPermission( let locations: readonly ToolLocation[] | undefined; if (tool.locations) { try { - locations = z - .array(ToolLocationSchema) - .parse(tool.locations(structuredClone(input))); + locations = toolLocations(tool, input); } catch (error) { host.emit({ type: "tool.locations_failed", @@ -91,8 +83,10 @@ export function requestPermission( }); } } - if (locations) + if (locations) { + grant.locations.set(call.id, locations); host.notifyTool({ ...identity, sessionUpdate: "tool_call_update", locations }); + } signal.throwIfAborted(); const permissionStartedAt = performance.now(); const permissionContext = { diff --git a/packages/core-agent/host/commands/tools.ts b/packages/core-agent/host/commands/tools.ts index 9a213591..cc7ffc56 100644 --- a/packages/core-agent/host/commands/tools.ts +++ b/packages/core-agent/host/commands/tools.ts @@ -14,7 +14,8 @@ import { Actor } from "../../fsm/fsm.ts"; import { diagnosticError } from "../../logging/index.ts"; import type { HostContext } from "../context.ts"; import type { ExecutionContext, HostToolOutcome } from "../host.ts"; -import { ToolLocationSchema, ToolOutputSchema } from "../ports.ts"; +import { ToolOutputSchema } from "../ports.ts"; +import { settledTool, toolAnnouncement, toolLocations } from "../tool-display.ts"; /** Largest text-media tool part inlined as text (the same 256 KiB as MCP and read_file results). */ const MAX_INLINE_TEXT = 256 * 1024; @@ -71,6 +72,10 @@ export function runTools( toolCallId: batchCommand.child.id, name: batchCommand.call.name, }; + // What the card shows, recorded with the call's outcome: the kind it was announced with and + // the locations resolved from its input, whether while permission was asked or here. + let locations = grant?.locations.get(batchCommand.call.id); + const shown = () => ({ kind: tool.kind ?? "other", ...(locations ? { locations } : {}) }); const refused = grant?.refused.get(batchCommand.call.id); host.emit({ type: "tool.admitted", @@ -92,17 +97,11 @@ export function runTools( toolName: batchCommand.call.name, ...(command.permission ? { permissionChildId: command.permission.id } : {}), }); - if (!host.closed) - host.notifyTool({ - ...identity, - sessionUpdate: "tool_call_update", - status: "failed", - rawOutput: { refused: true, reason: refused.message }, - }); const outcome: HostToolOutcome = { turnId, batchId: command.child.id, callId: batchCommand.call.id, + shown: shown(), result: { kind: "failed", error: failure(refused, { @@ -117,6 +116,7 @@ export function runTools( }), }, }; + if (!host.closed) host.notifyTool(settledTool(identity, outcome.result)); host.pendingTools.set(`${outcome.batchId}/${outcome.callId}`, { outcome, batch, @@ -127,16 +127,7 @@ export function runTools( }); break; } - if (!grant) - host.notifyTool({ - ...identity, - sessionUpdate: "tool_call", - title: batchCommand.call.name, - name: batchCommand.call.name, - kind: tool.kind ?? "other", - status: "pending", - rawInput: batchCommand.call.args, - }); + if (!grant) host.notifyTool(toolAnnouncement(identity, batchCommand.call, tool)); if (host.closed) break; let status = "pending"; @@ -164,9 +155,7 @@ export function runTools( const input = await tool.parseInput(raw); if (!host.closed && status === "pending" && tool.locations) { try { - const locations = z - .array(ToolLocationSchema) - .parse(tool.locations(structuredClone(input))); + locations = toolLocations(tool, input)!; host.emit({ type: "tool.locations_resolved", ...identity, @@ -238,6 +227,7 @@ export function runTools( batchId: command.child.id, callId: batchCommand.call.id, result, + shown: shown(), }; host.pendingTools.set(`${outcome.batchId}/${outcome.callId}`, { outcome, @@ -272,21 +262,15 @@ export function runTools( ...(state.status === "failed" ? { error: diagnosticError(state.error) } : {}), }); status = next; - host.notifyTool({ - ...identity, - sessionUpdate: "tool_call_update", - status: next, - ...(state.status === "succeeded" - ? { - rawOutput: state.value.text, - ...(state.value.parts ? { parts: state.value.parts } : {}), - } + host.notifyTool( + state.status === "succeeded" + ? settledTool(identity, { kind: "succeeded", value: state.value }) : state.status === "failed" - ? { rawOutput: { error: state.error.message } } + ? settledTool(identity, { kind: "failed", error: state.error }) : state.status === "cancelled" - ? { rawOutput: { error: "Tool cancelled" } } - : {}), - }); + ? settledTool(identity, { kind: "cancelled" }) + : { ...identity, sessionUpdate: "tool_call_update", status: next }, + ); }, ); break; diff --git a/packages/core-agent/host/context.ts b/packages/core-agent/host/context.ts index 337398ac..4d5ac5f5 100644 --- a/packages/core-agent/host/context.ts +++ b/packages/core-agent/host/context.ts @@ -21,6 +21,7 @@ import { type ExecutionBindings, type PermissionPort, type Tool, + type ToolLocation, } from "./ports.ts"; /** @@ -60,6 +61,8 @@ export type PermissionGrant = { refused: Map; pending: HostToolNotification[]; remembered: Map; + /** Locations shown for each call while permission was asked. */ + locations: Map; }; /** @@ -119,18 +122,7 @@ export function createHostContext( const emit = fanoutEffects(diagnosticsSubscriber(), bindings.effects); const remembered = new Map(); - const grants = new Map< - ActorId, - { - batchId: ActorId; - approved: boolean; - inputs: Map; - invalidInputs: Map; - refused: Map; - pending: HostToolNotification[]; - remembered: Map; - } - >(); + const grants = new Map(); const ctx: HostContext = { closed: false, diff --git a/packages/core-agent/host/host.ts b/packages/core-agent/host/host.ts index 2e28a35f..1f37a2da 100644 --- a/packages/core-agent/host/host.ts +++ b/packages/core-agent/host/host.ts @@ -15,6 +15,7 @@ import { type ExecutionBindings, type ToolKind, type ToolLocation, + type ToolShown, } from "./ports.ts"; /** @@ -29,6 +30,8 @@ export type HostToolOutcome = Readonly<{ callId: ToolCall["id"]; /** Raw result, before the configuration's `toolFailure` handling (applied on release). */ result: Result; + /** How the call's card was shown. */ + shown: ToolShown; }>; /** * A display update about one tool call, for UI cards. Non-authoritative display data. toolCallId is diff --git a/packages/core-agent/host/index.ts b/packages/core-agent/host/index.ts index 429abb9f..41c30bd6 100644 --- a/packages/core-agent/host/index.ts +++ b/packages/core-agent/host/index.ts @@ -1,2 +1,3 @@ export * from "./host.ts"; export * from "./ports.ts"; +export * from "./tool-display.ts"; diff --git a/packages/core-agent/host/ports.ts b/packages/core-agent/host/ports.ts index 34d2435a..69fb95d0 100644 --- a/packages/core-agent/host/ports.ts +++ b/packages/core-agent/host/ports.ts @@ -90,6 +90,20 @@ export const ToolLocationSchema = z /** A file a tool call reads or changes, for display; see {@link ToolLocationSchema}. */ export type ToolLocation = z.infer; +/** + * How a tool call's card was shown: the tool's kind and the locations resolved from its input, + * recorded with the call's outcome so a reopened session draws the same card whatever the tool is + * now. + */ +export const ToolShownSchema = z + .strictObject({ + kind: z.lazy(() => ToolKindSchema), + locations: z.array(ToolLocationSchema).readonly().optional(), + }) + .readonly(); + +export type ToolShown = z.infer; + /** * Identity of one live tool invocation, passed to {@link Tool.run}. * Ephemeral display identity for this live invocation; never journaled or an authorization grant. diff --git a/packages/core-agent/host/tool-display.ts b/packages/core-agent/host/tool-display.ts new file mode 100644 index 00000000..c0210669 --- /dev/null +++ b/packages/core-agent/host/tool-display.ts @@ -0,0 +1,85 @@ +import { z } from "zod"; + +import { toolOperationId, type ToolRunResult } from "../agent/tool-batch.ts"; +import { ref, type ActorId, type Result, type ToolCall } from "../agent/types.ts"; +import type { HostToolNotification } from "./host.ts"; +import { ToolLocationSchema, type Tool, type ToolKind, type ToolLocation } from "./ports.ts"; + +/** The ID a client knows call `callId` of tool batch `batchId` by: its `tool` operation's ID. */ +export const toolCallIdOf = (batchId: string, callId: string): ActorId => + ref("tool", toolOperationId(batchId, callId)).id; + +/** + * The fields every {@link HostToolNotification} about one call carries. `toolCallId` is the tool + * operation's ID, `/`. + */ +export type ToolIdentity = Readonly<{ + sessionId?: string; + turnId: ActorId; + batchId: ActorId; + callId: ToolCall["id"]; + toolCallId: ActorId; + name: string; +}>; + +/** + * The `tool_call` that announces a call, with the kind it is shown as: the tool's when announced + * live, the recorded one when a journal is drawn. Without one it is `other`, the protocol's default. + */ +export function toolAnnouncement( + identity: ToolIdentity, + call: Readonly<{ name: string; args: unknown }>, + shown: Readonly<{ kind?: ToolKind }> | undefined, +): HostToolNotification { + return { + ...identity, + sessionUpdate: "tool_call", + title: call.name, + name: call.name, + kind: shown?.kind ?? "other", + status: "pending", + rawInput: call.args, + }; +} + +/** + * The call's display locations from its tool's `locations` over the parsed input, or undefined when + * the tool declares none. + * + * @throws When `locations` throws or returns an invalid location. + */ +export function toolLocations( + tool: Pick, + input: unknown, +): readonly ToolLocation[] | undefined { + if (!tool.locations) return undefined; + return z.array(ToolLocationSchema).parse(tool.locations(structuredClone(input))); +} + +/** + * The last `tool_call_update` of a settled call, from its raw outcome: the output text and blob + * parts on success; `{ refused, reason }` for a refused permission; `{ error }` for any other + * failure; `{ error: "Tool cancelled" }` on cancellation. + */ +export function settledTool( + identity: ToolIdentity, + result: Result, +): HostToolNotification { + const settled = { ...identity, sessionUpdate: "tool_call_update" as const }; + if (result.kind === "succeeded") + return { + ...settled, + status: "completed", + rawOutput: result.value.text, + ...(result.value.parts ? { parts: result.value.parts } : {}), + }; + if (result.kind === "cancelled") + return { ...settled, status: "failed", rawOutput: { error: "Tool cancelled" } }; + if (result.error.classification === "permission_refused") + return { + ...settled, + status: "failed", + rawOutput: { refused: true, reason: result.error.message }, + }; + return { ...settled, status: "failed", rawOutput: { error: result.error.message } }; +} diff --git a/packages/core-agent/providers/types.ts b/packages/core-agent/providers/types.ts index ad5d5217..13e65b39 100644 --- a/packages/core-agent/providers/types.ts +++ b/packages/core-agent/providers/types.ts @@ -179,6 +179,20 @@ export const StreamDeltaSchema = z export type StreamDelta = z.infer; +/** + * What a client was sent of one step's stream: its thinking text, and, for a step that did not + * produce an answer (failed or cancelled), the answer text streamed before it stopped. A settled + * answer's text is its completion and is not repeated here. Empty when nothing streamed. + */ +export const ShownSchema = z + .strictObject({ + thinking: z.string().min(1).optional(), + text: z.string().min(1).optional(), + }) + .readonly(); + +export type Shown = z.infer; + export type StreamDeltaSink = (delta: StreamDelta) => unknown; export type StreamEvent = Readonly<{ event?: string; data: string }>; diff --git a/packages/core-agent/session/fixtures/expected-v2.json b/packages/core-agent/session/fixtures/expected-v2.json index 7d6a4d12..94bf977e 100644 --- a/packages/core-agent/session/fixtures/expected-v2.json +++ b/packages/core-agent/session/fixtures/expected-v2.json @@ -200,6 +200,7 @@ "event": { "type": "model_settled", "model": "example-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -381,6 +382,7 @@ "event": { "type": "model_settled", "model": "example-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -764,6 +766,7 @@ "event": { "type": "model_settled", "model": "example-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -836,6 +839,9 @@ "value": { "text": "{\"body\":\"Keep decisions pure; persist before releasing effects.\"}" } + }, + "shown": { + "kind": "read" } } }, @@ -878,6 +884,7 @@ "event": { "type": "model_settled", "model": "example-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/4" @@ -1261,6 +1268,7 @@ "event": { "type": "model_settled", "model": "example-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -1347,6 +1355,9 @@ "permissionChildId": "00000000-0000-4000-8000-000000000001/turn/1/2" } } + }, + "shown": { + "kind": "read" } } }, @@ -1389,6 +1400,7 @@ "event": { "type": "model_settled", "model": "example-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/4" @@ -1763,6 +1775,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -2078,6 +2091,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -2534,6 +2548,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -2615,6 +2630,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -2869,6 +2885,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -2950,6 +2967,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -3890,6 +3908,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -3952,6 +3971,9 @@ "message": "broken" } } + }, + "shown": { + "kind": "other" } } }, @@ -3971,6 +3993,9 @@ "value": { "text": "ok" } + }, + "shown": { + "kind": "other" } } }, @@ -4013,6 +4038,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -4268,6 +4294,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -4330,6 +4357,9 @@ "message": "broken" } } + }, + "shown": { + "kind": "other" } } }, @@ -4349,6 +4379,9 @@ "value": { "text": "ok" } + }, + "shown": { + "kind": "other" } } }, @@ -4391,6 +4424,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -4803,6 +4837,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -4851,6 +4886,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } }, @@ -4956,6 +4994,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -5227,6 +5266,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -5275,6 +5315,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } }, @@ -5380,6 +5423,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -5825,6 +5869,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -6129,6 +6174,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000007/turn/1/1" @@ -6160,6 +6206,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000007/turn/1/2" @@ -6403,6 +6450,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000007/turn/1/1" @@ -6434,6 +6482,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000007/turn/1/2" @@ -6836,6 +6885,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -6920,6 +6970,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -7161,6 +7212,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -7245,6 +7297,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -7545,6 +7598,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -7770,6 +7824,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -8875,6 +8930,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -8916,6 +8972,9 @@ "value": { "text": "result-0" } + }, + "shown": { + "kind": "other" } } }, @@ -8975,6 +9034,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -9016,6 +9076,9 @@ "value": { "text": "result-1" } + }, + "shown": { + "kind": "other" } } }, @@ -9075,6 +9138,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/5" @@ -9203,6 +9267,7 @@ "type": "model_settled", "provider": "google-generate@1", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -9551,6 +9616,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -9592,6 +9658,9 @@ "value": { "text": "result-0" } + }, + "shown": { + "kind": "other" } } }, @@ -9651,6 +9720,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -9692,6 +9762,9 @@ "value": { "text": "result-1" } + }, + "shown": { + "kind": "other" } } }, @@ -9751,6 +9824,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/5" @@ -10304,6 +10378,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000011/turn/2/1" @@ -10562,6 +10637,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000016/turn/1/1" @@ -11119,6 +11195,7 @@ "type": "model_settled", "provider": "openai-chat@1", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -11425,6 +11502,7 @@ "type": "model_settled", "provider": "openai-chat@1", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -11720,6 +11798,7 @@ "type": "model_settled", "provider": "openai-chat@1", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000005/turn/2/1" @@ -11984,6 +12063,7 @@ "type": "model_settled", "provider": "openai-chat@1", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000010/turn/1/1" @@ -12790,6 +12870,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -12838,6 +12919,9 @@ "value": { "text": "result-0-0" } + }, + "shown": { + "kind": "other" } } }, @@ -12857,6 +12941,9 @@ "value": { "text": "result-0-1" } + }, + "shown": { + "kind": "other" } } }, @@ -12939,6 +13026,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -12987,6 +13075,9 @@ "value": { "text": "result-1-0" } + }, + "shown": { + "kind": "other" } } }, @@ -13006,6 +13097,9 @@ "value": { "text": "result-1-1" } + }, + "shown": { + "kind": "other" } } }, @@ -13064,6 +13158,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/5" @@ -13504,6 +13599,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -13552,6 +13648,9 @@ "value": { "text": "result-0-0" } + }, + "shown": { + "kind": "other" } } }, @@ -13571,6 +13670,9 @@ "value": { "text": "result-0-1" } + }, + "shown": { + "kind": "other" } } }, @@ -13653,6 +13755,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -13701,6 +13804,9 @@ "value": { "text": "result-1-0" } + }, + "shown": { + "kind": "other" } } }, @@ -13720,6 +13826,9 @@ "value": { "text": "result-1-1" } + }, + "shown": { + "kind": "other" } } }, @@ -13778,6 +13887,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/5" @@ -14470,6 +14580,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -14511,6 +14622,9 @@ "value": { "text": "result-0" } + }, + "shown": { + "kind": "other" } } }, @@ -14571,6 +14685,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -14612,6 +14727,9 @@ "value": { "text": "result-1" } + }, + "shown": { + "kind": "other" } } }, @@ -14672,6 +14790,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/5" @@ -15020,6 +15139,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -15061,6 +15181,9 @@ "value": { "text": "result-0" } + }, + "shown": { + "kind": "other" } } }, @@ -15121,6 +15244,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -15162,6 +15286,9 @@ "value": { "text": "result-1" } + }, + "shown": { + "kind": "other" } } }, @@ -15222,6 +15349,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/5" @@ -15931,6 +16059,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -16037,6 +16166,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -16141,6 +16271,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/3/1" @@ -16483,6 +16614,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -16589,6 +16721,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -16693,6 +16826,7 @@ ] } }, + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/3/1" @@ -17185,6 +17319,7 @@ "type": "model_settled", "provider": "openai-chat@2", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -17253,6 +17388,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -17296,6 +17434,7 @@ "type": "model_settled", "provider": "openai-chat@2", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -17593,6 +17732,7 @@ "type": "model_settled", "provider": "openai-chat@2", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -17661,6 +17801,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -17704,6 +17847,7 @@ "type": "model_settled", "provider": "openai-chat@2", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -18241,6 +18385,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -18312,6 +18459,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -18371,6 +18521,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -18734,6 +18887,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -18805,6 +18961,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -18864,6 +19023,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -19469,6 +19631,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -19537,6 +19702,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -19602,6 +19770,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -19995,6 +20166,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -20063,6 +20237,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -20128,6 +20305,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -20704,6 +20884,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -20772,6 +20955,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -20832,6 +21018,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -21190,6 +21379,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -21258,6 +21450,9 @@ "value": { "text": "Hello 🌍" } + }, + "shown": { + "kind": "other" } } }, @@ -21318,6 +21513,9 @@ ] } }, + "shown": { + "thinking": "Considering" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -21772,6 +21970,9 @@ "type": "model_settled", "provider": "openai-chat@2", "model": "test-model", + "shown": { + "text": "Hello 🌍" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -22075,6 +22276,9 @@ "type": "model_settled", "provider": "openai-chat@2", "model": "test-model", + "shown": { + "text": "Hello 🌍" + }, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -22618,6 +22822,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -22701,6 +22906,9 @@ "value": { "text": "one" } + }, + "shown": { + "kind": "other" } } }, @@ -22720,6 +22928,9 @@ "value": { "text": "two" } + }, + "shown": { + "kind": "other" } } }, @@ -22762,6 +22973,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/4" @@ -23019,6 +23231,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -23102,6 +23315,9 @@ "value": { "text": "one" } + }, + "shown": { + "kind": "other" } } }, @@ -23121,6 +23337,9 @@ "value": { "text": "two" } + }, + "shown": { + "kind": "other" } } }, @@ -23163,6 +23382,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/4" @@ -23656,6 +23876,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -23753,6 +23974,9 @@ "permissionChildId": "00000000-0000-4000-8000-000000000001/turn/1/2" } } + }, + "shown": { + "kind": "other" } } }, @@ -23772,6 +23996,9 @@ "value": { "text": "one" } + }, + "shown": { + "kind": "other" } } }, @@ -23814,6 +24041,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/4" @@ -24071,6 +24299,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -24168,6 +24397,9 @@ "permissionChildId": "00000000-0000-4000-8000-000000000001/turn/1/2" } } + }, + "shown": { + "kind": "other" } } }, @@ -24187,6 +24419,9 @@ "value": { "text": "one" } + }, + "shown": { + "kind": "other" } } }, @@ -24229,6 +24464,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/4" @@ -24617,6 +24853,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -25030,6 +25267,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -25288,6 +25526,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" diff --git a/packages/core-agent/session/fixtures/expected.json b/packages/core-agent/session/fixtures/expected.json index c7dbb7ce..d37536ec 100644 --- a/packages/core-agent/session/fixtures/expected.json +++ b/packages/core-agent/session/fixtures/expected.json @@ -354,6 +354,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -414,6 +415,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -655,6 +657,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -715,6 +718,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -1047,6 +1051,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -1284,6 +1289,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -1707,6 +1713,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -1755,6 +1762,9 @@ "value": { "text": "one" } + }, + "shown": { + "kind": "other" } } }, @@ -1774,6 +1784,9 @@ "value": { "text": "two" } + }, + "shown": { + "kind": "other" } } }, @@ -1816,6 +1829,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/3" @@ -2176,6 +2190,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -2224,6 +2239,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } }, @@ -2514,6 +2532,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -2562,6 +2581,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } }, @@ -2987,6 +3009,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -3280,6 +3303,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000005/turn/2/1" @@ -3554,6 +3578,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000005/turn/2/1" @@ -3970,6 +3995,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -4030,6 +4056,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/2/1" @@ -4586,6 +4613,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -4850,6 +4878,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000005/turn/1/1" @@ -5736,6 +5765,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -5767,6 +5797,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/2" @@ -6008,6 +6039,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -6039,6 +6071,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/2" @@ -6339,6 +6372,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -6662,6 +6696,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -6924,6 +6959,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -7314,6 +7350,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } ], @@ -7419,6 +7458,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -7467,6 +7507,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } } @@ -7699,6 +7742,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -7747,6 +7791,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } }, @@ -8020,6 +8067,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -8068,6 +8116,9 @@ "value": { "text": "fast" } + }, + "shown": { + "kind": "other" } } }, @@ -9068,6 +9119,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -9293,6 +9345,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/1" @@ -10125,6 +10178,7 @@ "event": { "type": "model_settled", "model": "test-model", + "shown": {}, "child": { "kind": "completion", "id": "00000000-0000-4000-8000-000000000001/turn/1/2" diff --git a/packages/core-agent/session/journal/domain-event.ts b/packages/core-agent/session/journal/domain-event.ts index 81269b6e..44d99251 100644 --- a/packages/core-agent/session/journal/domain-event.ts +++ b/packages/core-agent/session/journal/domain-event.ts @@ -5,14 +5,13 @@ import { partialResults } from "./shared.ts"; import type { Fold, JournalState } from "./state.ts"; /** - * Converts a {@link WireEvent} into the richer event the turn machine reads: - * `model_settled` gets a `permissionRequired` flag derived fresh from the - * policy in force (never stored, so there is nothing to compare against on load), and - * `batch_settled` gets its results assembled from the batch's committed `tool` records (never - * stored on the wire either). A failed dispatch of a turn's child operation becomes a `failed` + * Converts a {@link WireEvent} into the event the turn machine reads. `model_settled` gains + * `permissionRequired`, computed here from `state.policy`: the step's record does not hold whether + * permission was asked. `batch_settled` gains its results, read from the batch's `tool` records, + * which hold each call's outcome. A failed dispatch of a turn's child operation becomes a `failed` * child event. * - * @throws Error for a failed branch reply, which has no journaled form. + * @throws Error for a failed branch reply, which has no journal record. */ export function domainEvent(state: JournalState, event: WireEvent, fold: Fold): ConversationEvent { if (event.type !== "child") return event; @@ -41,8 +40,7 @@ export function domainEvent(state: JournalState, event: WireEvent, fold: Fold): ) throw new Error("Unpermitted tool"); } - // The permission route is always derived from the policy in force, both when staging and on - // load; it is never stored, so there is nothing for a stored copy to disagree with. + // Computed from the policy in force at this point in the journal, when staging and on load. const permissionRequired = state.policy?.permissions === "ask" && child.result.kind === "succeeded" && @@ -53,8 +51,7 @@ export function domainEvent(state: JournalState, event: WireEvent, fold: Fold): }; } if (child.type !== "batch_settled") return { ...event, event: child }; - // Individual results are never stored on the wire; both staging and load assemble them from the - // batch's committed `tool` records. + // Each call's outcome is in its `tool` record; the batch's results are read from those. const results = partialResults(state); if (child.outcome.kind !== "succeeded") return { ...event, event: { ...child, outcome: { ...child.outcome, results } } }; diff --git a/packages/core-agent/session/streaming-runtime.test.ts b/packages/core-agent/session/streaming-runtime.test.ts index 586c43e4..23cf9bcd 100644 --- a/packages/core-agent/session/streaming-runtime.test.ts +++ b/packages/core-agent/session/streaming-runtime.test.ts @@ -103,6 +103,17 @@ for (const profile of streamingProfiles) { record.body.event.event.type === "model_settled", ); expect(settlements).toHaveLength(3); + // Each step records the thinking its stream showed; a settled answer's text is its completion. + for (const { body } of settlements) { + if (body.kind !== "event" || body.event.type !== "child") continue; + const event = body.event.event; + if (event.type !== "model_settled") continue; + const thinking = updates + .filter((update) => update.completionId === event.child.id) + .map((update) => update.thinking ?? "") + .join(""); + expect(event.shown).toEqual(thinking ? { thinking } : {}); + } expect(journalJSONL(durable)).not.toContain('"completion_update"'); if (profile.id !== "openai-chat@2") { expect(durable.continuations).toHaveLength(3); @@ -115,6 +126,10 @@ for (const profile of streamingProfiles) { expect(updates).toHaveLength(count); const fork = await session.fork(); await fork.input("Next").settled; + // The thinking a step showed is recorded for display only: without a provider continuation to + // carry it, the model never sees it again. + if (profile.id === "openai-chat@2") + expect(JSON.stringify(requests.at(-1))).not.toContain("Considering"); expect(updates.at(-1)?.sessionId).toBe(fork.snapshot.durable.conversation.sessionId); expect(updates.at(-1)?.status).toBe("completed"); await Promise.all([session, restored, fork].map((runtime) => runtime.close())); @@ -159,6 +174,22 @@ for (const profile of streamingProfiles) { }); expect(updates.at(-1)?.status).toBe("failed"); expect(updates.some((event) => event.status === "completed")).toBe(false); + // A step that failed mid-stream still records what its stream showed, answer text included. + const shown = (field: "thinking" | "text") => + updates.map((event) => event[field] ?? "").join(""); + expect(settlements[0]).toMatchObject({ + body: { + event: { + event: { + shown: { + ...(shown("thinking") ? { thinking: shown("thinking") } : {}), + ...(shown("text") ? { text: shown("text") } : {}), + }, + }, + }, + }, + }); + expect(shown("thinking")).not.toBe(""); await session.close(); }); } diff --git a/packages/core-agent/session/types.ts b/packages/core-agent/session/types.ts index 908111fc..a85603ca 100644 --- a/packages/core-agent/session/types.ts +++ b/packages/core-agent/session/types.ts @@ -17,7 +17,8 @@ import { TurnRecordSchema, } from "../agent/types.ts"; import { PolicySchema, PolicyVersionSchema } from "../policy/policy.ts"; -import { ContinuationSchema } from "../providers/types.ts"; +import { ToolShownSchema } from "../host/ports.ts"; +import { ContinuationSchema, ShownSchema } from "../providers/types.ts"; import { CompletionUsageSchema, type CompletionUsage } from "../providers/usage.ts"; import { AppendIdSchema, RevisionSchema } from "./persistence.ts"; @@ -199,6 +200,12 @@ export const WireEventSchema = z.discriminatedUnion("type", [ /** Application model name the step was made with, whether or not it produced output. */ model: z.string().min(1), continuation: ContinuationSchema.optional(), + /** + * What the client was shown while the step streamed; see {@link ShownSchema}. Every step + * records it, even when empty, so its absence means the step was recorded without it and a + * reader cannot know what was shown. + */ + shown: ShownSchema.optional(), child: child("completion"), result: result(CompletionSchema.brand<"AdmittedCompletion">()), }), @@ -299,6 +306,8 @@ export const BodySchema = z.discriminatedUnion("kind", [ callId: ToolCallIdSchema, /** Raw outcome; see {@link ToolRunOutcomeSchema}. */ result: ToolRunOutcomeSchema, + /** How the call's card was shown. Absent on calls recorded without it. */ + shown: ToolShownSchema.optional(), }), z.strictObject({ kind: z.literal("effect"), diff --git a/packages/core-agent/session/views.ts b/packages/core-agent/session/views.ts index 43f7e362..d9a27ee6 100644 --- a/packages/core-agent/session/views.ts +++ b/packages/core-agent/session/views.ts @@ -51,7 +51,8 @@ export type ConversationView = Readonly<{ live: readonly MessageView[]; }>; -function projectBlob(ref: BlobRef): ViewBlob { +/** A stored blob as a view shows it, with its `blob://` URI. */ +export function projectBlob(ref: BlobRef): ViewBlob { return { id: ref.id, media: ref.media,