|
|
@@ -52,6 +52,7 @@ import { AgentV2 } from "@opencode-ai/core/agent"
|
|
|
import { Config } from "@opencode-ai/core/config"
|
|
|
import { ConfigCompaction } from "@opencode-ai/core/config/compaction"
|
|
|
import { Tool } from "@opencode-ai/core/tool/tool"
|
|
|
+import { ToolHooks } from "@opencode-ai/core/tool/hooks"
|
|
|
import {
|
|
|
InstructionStateTable,
|
|
|
SessionPendingTable,
|
|
|
@@ -238,43 +239,45 @@ const permission = Layer.succeed(
|
|
|
)
|
|
|
const echo = Layer.effectDiscard(
|
|
|
ToolRegistry.Service.use((registry) =>
|
|
|
- registry.register({
|
|
|
- echo: Tool.make({
|
|
|
- description: "Echo text",
|
|
|
- input: Schema.Struct({ text: Schema.String }),
|
|
|
- output: Schema.Struct({ text: Schema.String }),
|
|
|
- toModelOutput: ({ output }) => [{ type: "text", text: output.text }],
|
|
|
- execute: ({ text }, context) =>
|
|
|
- Effect.gen(function* () {
|
|
|
- authorizations.push(context)
|
|
|
- executions.push(text)
|
|
|
- activeToolExecutions++
|
|
|
- maxActiveToolExecutions = Math.max(maxActiveToolExecutions, activeToolExecutions)
|
|
|
- if (activeToolExecutions === toolExecutionsReady && toolExecutionsStarted) {
|
|
|
- yield* Deferred.succeed(toolExecutionsStarted, undefined)
|
|
|
- }
|
|
|
- if (toolExecutionGate) yield* Deferred.await(toolExecutionGate)
|
|
|
- return { text }
|
|
|
- }).pipe(Effect.ensuring(Effect.sync(() => activeToolExecutions--))),
|
|
|
- }),
|
|
|
- defect: Tool.make({
|
|
|
- description: "Fail unexpectedly",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({}),
|
|
|
- execute: () =>
|
|
|
- (toolExecutionGate ? Deferred.await(toolExecutionGate) : Effect.void).pipe(
|
|
|
- Effect.andThen(Effect.die("unexpected tool defect")),
|
|
|
- ),
|
|
|
- }),
|
|
|
- // BigInt output with no model content forces ToolOutputStore.bound onto its
|
|
|
- // JSON.stringify encode path, which fails with a typed StorageError.
|
|
|
- storefail: Tool.make({
|
|
|
- description: "Produce output that cannot be persisted",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Any,
|
|
|
- execute: () => Effect.succeed({ big: 1n }),
|
|
|
- }),
|
|
|
- }, { codemode: false }),
|
|
|
+ registry.register(
|
|
|
+ {
|
|
|
+ echo: Tool.make({
|
|
|
+ description: "Echo text",
|
|
|
+ input: Schema.Struct({ text: Schema.String }),
|
|
|
+ output: Schema.Struct({ text: Schema.String }),
|
|
|
+ execute: ({ text }, context) =>
|
|
|
+ Effect.gen(function* () {
|
|
|
+ authorizations.push(context)
|
|
|
+ executions.push(text)
|
|
|
+ activeToolExecutions++
|
|
|
+ maxActiveToolExecutions = Math.max(maxActiveToolExecutions, activeToolExecutions)
|
|
|
+ if (activeToolExecutions === toolExecutionsReady && toolExecutionsStarted) {
|
|
|
+ yield* Deferred.succeed(toolExecutionsStarted, undefined)
|
|
|
+ }
|
|
|
+ if (toolExecutionGate) yield* Deferred.await(toolExecutionGate)
|
|
|
+ return { output: { text }, content: text }
|
|
|
+ }).pipe(Effect.ensuring(Effect.sync(() => activeToolExecutions--))),
|
|
|
+ }),
|
|
|
+ defect: Tool.make({
|
|
|
+ description: "Fail unexpectedly",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: () =>
|
|
|
+ (toolExecutionGate ? Deferred.await(toolExecutionGate) : Effect.void).pipe(
|
|
|
+ Effect.andThen(Effect.die("unexpected tool defect")),
|
|
|
+ ),
|
|
|
+ }),
|
|
|
+ // The wrapped ToolOutputStore below fails bound for this call ID with a
|
|
|
+ // typed StorageError, exercising the infrastructure failure channel.
|
|
|
+ storefail: Tool.make({
|
|
|
+ description: "Produce output that cannot be persisted",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: () => Effect.succeed({ output: {} }),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ ),
|
|
|
),
|
|
|
)
|
|
|
const echoNode = makeLocationNode({ name: "test/session-runner-tools", layer: echo, deps: [ToolRegistry.node] })
|
|
|
@@ -379,6 +382,15 @@ const promptCatalog = Layer.mock(Catalog.Service, {
|
|
|
small: () => Effect.succeed(undefined),
|
|
|
},
|
|
|
})
|
|
|
+// Pass-through bounding that fails "call-storefail" with a typed StorageError so
|
|
|
+// runner tests can exercise the infrastructure failure channel deterministically.
|
|
|
+const toolOutputStore = Layer.mock(ToolOutputStore.Service, {
|
|
|
+ limits: () => Effect.succeed({ maxLines: ToolOutputStore.MAX_LINES, maxBytes: ToolOutputStore.MAX_BYTES }),
|
|
|
+ bound: (input) =>
|
|
|
+ input.callID === "call-storefail"
|
|
|
+ ? Effect.fail(new ToolOutputStore.StorageError({ operation: "write", cause: new Error("disk full") }))
|
|
|
+ : Effect.succeed({ content: input.content, outputPaths: [] }),
|
|
|
+})
|
|
|
const runnerLayer = AppNodeBuilder.build(SessionRunnerLLM.node, [
|
|
|
[Snapshot.node, Snapshot.noopLayer],
|
|
|
[LayerNodePlatform.llmClient, client],
|
|
|
@@ -391,7 +403,7 @@ const runnerLayer = AppNodeBuilder.build(SessionRunnerLLM.node, [
|
|
|
[PermissionV2.node, permission],
|
|
|
[Config.node, config],
|
|
|
[McpInstructions.node, mcpInstructions],
|
|
|
- [ToolOutputStore.node, ToolOutputStore.nodeWithoutConfig],
|
|
|
+ [ToolOutputStore.node, toolOutputStore],
|
|
|
[PluginSupervisor.node, pluginSupervisor],
|
|
|
])
|
|
|
const execution = Layer.effect(
|
|
|
@@ -422,6 +434,7 @@ const it = testEffect(
|
|
|
Catalog.node,
|
|
|
ToolRegistry.node,
|
|
|
ToolRegistry.toolsNode,
|
|
|
+ ToolHooks.node,
|
|
|
PluginHooks.node,
|
|
|
echoNode,
|
|
|
SessionRunnerModel.node,
|
|
|
@@ -449,7 +462,7 @@ const it = testEffect(
|
|
|
[Snapshot.node, Snapshot.noopLayer],
|
|
|
[SessionExecution.node, execution],
|
|
|
[Config.node, config],
|
|
|
- [ToolOutputStore.node, ToolOutputStore.nodeWithoutConfig],
|
|
|
+ [ToolOutputStore.node, toolOutputStore],
|
|
|
[PluginSupervisor.node, pluginSupervisor],
|
|
|
],
|
|
|
),
|
|
|
@@ -586,8 +599,8 @@ const recordedStepSettlementEvents = (id: SessionV2.ID, assistantMessageID: Sess
|
|
|
const settlementTypes = new Set([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.success.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.success.2",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.ended.1",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
@@ -827,12 +840,26 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
- expect(requests).toHaveLength(1)
|
|
|
+ // A hook-removed call fails independently and continues while step allowance remains.
|
|
|
+ expect(requests).toHaveLength(2)
|
|
|
expect(requests[0]?.system.map((part) => part.text)).toEqual(["Hooked system"])
|
|
|
expect(requests[0]?.messages).toEqual([Message.user("Hooked message")])
|
|
|
expect(requests[0]?.tools.map((tool) => tool.name)).not.toContain("echo")
|
|
|
expect(requests[0]?.tools.map((tool) => tool.name)).not.toContain("unregistered")
|
|
|
expect(executions).toEqual([])
|
|
|
+ expect(yield* session.context(sessionID)).toMatchObject([
|
|
|
+ { type: "user", text: "Original message" },
|
|
|
+ {
|
|
|
+ type: "assistant",
|
|
|
+ content: [
|
|
|
+ {
|
|
|
+ type: "tool",
|
|
|
+ id: "call-removed",
|
|
|
+ state: { status: "error", error: { type: "tool.unknown" } },
|
|
|
+ },
|
|
|
+ ],
|
|
|
+ },
|
|
|
+ ])
|
|
|
}),
|
|
|
)
|
|
|
|
|
|
@@ -841,19 +868,22 @@ describe("SessionRunnerLLM", () => {
|
|
|
const session = yield* setup
|
|
|
const registry = yield* ToolRegistry.Service
|
|
|
const contexts: Tool.Context[] = []
|
|
|
- yield* registry.register({
|
|
|
- location_context: Tool.make({
|
|
|
- description: "Read application context",
|
|
|
- input: Schema.Struct({ query: Schema.String }),
|
|
|
- output: Schema.Struct({ answer: Schema.String }),
|
|
|
- execute: ({ query }, context) =>
|
|
|
- Effect.gen(function* () {
|
|
|
- contexts.push(context)
|
|
|
- yield* context.progress({ structured: { phase: "reading" } })
|
|
|
- return { answer: query.toUpperCase() }
|
|
|
- }),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ location_context: Tool.make({
|
|
|
+ description: "Read application context",
|
|
|
+ input: Schema.Struct({ query: Schema.String }),
|
|
|
+ output: Schema.Struct({ answer: Schema.String }),
|
|
|
+ execute: ({ query }, context) =>
|
|
|
+ Effect.gen(function* () {
|
|
|
+ contexts.push(context)
|
|
|
+ yield* context.progress({ phase: "reading" })
|
|
|
+ return { output: { answer: query.toUpperCase() } }
|
|
|
+ }),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* admit(session, "Use application context")
|
|
|
responses = [reply.tool("call-location", "location_context", { query: "hello" }), []]
|
|
|
const events = yield* EventV2.Service
|
|
|
@@ -876,7 +906,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
progress: expect.any(Function),
|
|
|
},
|
|
|
])
|
|
|
- expect(Array.from(yield* Fiber.join(progressFiber))[0]?.data.structured).toEqual({ phase: "reading" })
|
|
|
+ expect(Array.from(yield* Fiber.join(progressFiber))[0]?.data.metadata).toEqual({ phase: "reading" })
|
|
|
expect(yield* session.context(sessionID)).toMatchObject([
|
|
|
{ type: "user", text: "Use application context" },
|
|
|
{
|
|
|
@@ -885,7 +915,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "tool",
|
|
|
id: "call-location",
|
|
|
- state: { status: "completed", structured: { answer: "HELLO" } },
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: '{"answer":"HELLO"}' }] },
|
|
|
},
|
|
|
],
|
|
|
},
|
|
|
@@ -893,25 +923,29 @@ describe("SessionRunnerLLM", () => {
|
|
|
}),
|
|
|
)
|
|
|
|
|
|
- it.effect("persists the latest partial snapshot when a tool fails", () =>
|
|
|
+ it.effect("prefers failure outcome metadata over retained progress", () =>
|
|
|
Effect.gen(function* () {
|
|
|
const session = yield* setup
|
|
|
const registry = yield* ToolRegistry.Service
|
|
|
- yield* registry.register({
|
|
|
- failing_progress: Tool.make({
|
|
|
- description: "Report progress and fail",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({}),
|
|
|
- execute: (_, context) =>
|
|
|
- Effect.gen(function* () {
|
|
|
- yield* context.progress({
|
|
|
- structured: { phase: "running" },
|
|
|
- content: [{ type: "text", text: "before failure" }],
|
|
|
- })
|
|
|
- return yield* new ToolFailure({ message: "failed after progress" })
|
|
|
- }),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ const hooks = yield* ToolHooks.Service
|
|
|
+ yield* hooks.hook.after((event) => {
|
|
|
+ if (event.status === "error") event.metadata = { phase: "failed" }
|
|
|
+ })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ failing_progress: Tool.make({
|
|
|
+ description: "Report progress and fail",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: (_, context) =>
|
|
|
+ Effect.gen(function* () {
|
|
|
+ yield* context.progress({ phase: "running" })
|
|
|
+ return yield* new ToolFailure({ message: "failed after progress" })
|
|
|
+ }),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* admit(session, "Run failing progress")
|
|
|
responses = [reply.tool("call-failing-progress", "failing_progress", {}), reply.stop()]
|
|
|
|
|
|
@@ -927,8 +961,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
id: "call-failing-progress",
|
|
|
state: {
|
|
|
status: "error",
|
|
|
- structured: { phase: "running" },
|
|
|
- content: [{ type: "text", text: "before failure" }],
|
|
|
+ metadata: { phase: "failed" },
|
|
|
error: { message: "failed after progress" },
|
|
|
},
|
|
|
},
|
|
|
@@ -946,14 +979,20 @@ describe("SessionRunnerLLM", () => {
|
|
|
const scope = yield* Scope.make()
|
|
|
const executions: string[] = []
|
|
|
yield* registry
|
|
|
- .register({
|
|
|
- reloaded: Tool.make({
|
|
|
- description: "Record the advertised tool",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({ value: Schema.String }),
|
|
|
- execute: () => Effect.sync(() => executions.push("advertised")).pipe(Effect.as({ value: "advertised" })),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ .register(
|
|
|
+ {
|
|
|
+ reloaded: Tool.make({
|
|
|
+ description: "Record the advertised tool",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({ value: Schema.String }),
|
|
|
+ execute: () =>
|
|
|
+ Effect.sync(() => executions.push("advertised")).pipe(
|
|
|
+ Effect.as({ output: { value: "advertised" } }),
|
|
|
+ ),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
.pipe(Scope.provide(scope))
|
|
|
yield* admit(session, "Use the reloaded tool")
|
|
|
responses = [
|
|
|
@@ -971,14 +1010,20 @@ describe("SessionRunnerLLM", () => {
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
yield* Scope.close(scope, Exit.void)
|
|
|
- yield* registry.register({
|
|
|
- reloaded: Tool.make({
|
|
|
- description: "Record the replacement tool",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({ value: Schema.String }),
|
|
|
- execute: () => Effect.sync(() => executions.push("replacement")).pipe(Effect.as({ value: "replacement" })),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ reloaded: Tool.make({
|
|
|
+ description: "Record the replacement tool",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({ value: Schema.String }),
|
|
|
+ execute: () =>
|
|
|
+ Effect.sync(() => executions.push("replacement")).pipe(
|
|
|
+ Effect.as({ output: { value: "replacement" } }),
|
|
|
+ ),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
yield* Fiber.join(run)
|
|
|
|
|
|
@@ -991,7 +1036,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "tool",
|
|
|
id: "call-reloaded",
|
|
|
- state: { status: "completed", structured: { value: "advertised" } },
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: '{"value":"advertised"}' }] },
|
|
|
},
|
|
|
],
|
|
|
},
|
|
|
@@ -2377,7 +2422,6 @@ describe("SessionRunnerLLM", () => {
|
|
|
state: {
|
|
|
status: "completed",
|
|
|
input: { query: "hello" },
|
|
|
- structured: {},
|
|
|
content: [
|
|
|
{ type: "text", text: "Hello" },
|
|
|
{ type: "file", mime: "image/png", uri: "data:image/png;base64,aGVsbG8=", name: "hello.png" },
|
|
|
@@ -2417,7 +2461,6 @@ describe("SessionRunnerLLM", () => {
|
|
|
state: {
|
|
|
status: "completed",
|
|
|
input: { text: "hello" },
|
|
|
- structured: { text: "hello" },
|
|
|
content: [{ type: "text", text: "hello" }],
|
|
|
},
|
|
|
},
|
|
|
@@ -2429,7 +2472,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, assistant.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.success.1",
|
|
|
+ "session.tool.success.2",
|
|
|
"session.step.ended.1",
|
|
|
])
|
|
|
}),
|
|
|
@@ -2581,7 +2624,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
type: "tool-result",
|
|
|
id: "hosted-search",
|
|
|
name: "web_search",
|
|
|
- result: { type: "json", value: [{ title: "Effect" }] },
|
|
|
+ // The generic replay result derives from canonical stored content.
|
|
|
+ result: { type: "text", value: '[{"title":"Effect"}]' },
|
|
|
providerExecuted: true,
|
|
|
providerMetadata: { openai: { blockType: "web_search_tool_result" } },
|
|
|
},
|
|
|
@@ -2667,7 +2711,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "tool",
|
|
|
id: "tool_0",
|
|
|
- state: { status: "completed", structured: { text: "first" }, content: [{ type: "text", text: "first" }] },
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: "first" }] },
|
|
|
},
|
|
|
],
|
|
|
},
|
|
|
@@ -2677,11 +2721,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "tool",
|
|
|
id: "tool_0",
|
|
|
- state: {
|
|
|
- status: "completed",
|
|
|
- structured: { text: "second" },
|
|
|
- content: [{ type: "text", text: "second" }],
|
|
|
- },
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: "second" }] },
|
|
|
},
|
|
|
],
|
|
|
},
|
|
|
@@ -2697,7 +2737,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "tool",
|
|
|
id: "tool_0",
|
|
|
- state: { status: "completed", structured: { text: "first" }, content: [{ type: "text", text: "first" }] },
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: "first" }] },
|
|
|
},
|
|
|
],
|
|
|
},
|
|
|
@@ -2707,11 +2747,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "tool",
|
|
|
id: "tool_0",
|
|
|
- state: {
|
|
|
- status: "completed",
|
|
|
- structured: { text: "second" },
|
|
|
- content: [{ type: "text", text: "second" }],
|
|
|
- },
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: "second" }] },
|
|
|
},
|
|
|
],
|
|
|
},
|
|
|
@@ -3404,7 +3440,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, assistant.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.ended.1",
|
|
|
])
|
|
|
}),
|
|
|
@@ -3414,17 +3450,20 @@ describe("SessionRunnerLLM", () => {
|
|
|
Effect.gen(function* () {
|
|
|
const session = yield* setup
|
|
|
const registry = yield* ToolRegistry.Service
|
|
|
- yield* registry.register({
|
|
|
- blocked: Tool.make({
|
|
|
- description: "Fail because policy blocked execution",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({}),
|
|
|
- execute: () =>
|
|
|
- Effect.fail(new PermissionV2.BlockedError({ rules: [], permission: "blocked", resources: ["*"] })).pipe(
|
|
|
- Effect.mapError(() => new Tool.Failure({ message: "Permission blocked" })),
|
|
|
- ),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ blocked: Tool.make({
|
|
|
+ description: "Fail because policy blocked execution",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: () =>
|
|
|
+ Effect.fail(new PermissionV2.BlockedError({ rules: [], permission: "blocked", resources: ["*"] })).pipe(
|
|
|
+ Effect.mapError(() => new Tool.Failure({ message: "Permission blocked" })),
|
|
|
+ ),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* admit(session, "Call blocked")
|
|
|
|
|
|
responses = [reply.tool("call-blocked", "blocked", {}), reply.stop()]
|
|
|
@@ -3449,14 +3488,17 @@ describe("SessionRunnerLLM", () => {
|
|
|
Effect.gen(function* () {
|
|
|
const session = yield* setup
|
|
|
const registry = yield* ToolRegistry.Service
|
|
|
- yield* registry.register({
|
|
|
- declined: Tool.make({
|
|
|
- description: "Fail because the user declined approval",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({}),
|
|
|
- execute: () => Effect.die(new PermissionV2.DeclinedError()),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ declined: Tool.make({
|
|
|
+ description: "Fail because the user declined approval",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: () => Effect.die(new PermissionV2.DeclinedError()),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* admit(session, "Call declined")
|
|
|
|
|
|
response = reply.tool("call-declined", "declined", {})
|
|
|
@@ -3486,17 +3528,20 @@ describe("SessionRunnerLLM", () => {
|
|
|
Effect.gen(function* () {
|
|
|
const session = yield* setup
|
|
|
const registry = yield* ToolRegistry.Service
|
|
|
- yield* registry.register({
|
|
|
- corrected: Tool.make({
|
|
|
- description: "Fail with user correction feedback",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({}),
|
|
|
- execute: () =>
|
|
|
- Effect.fail(new PermissionV2.CorrectedError({ feedback: "Use another tool" })).pipe(
|
|
|
- Effect.mapError(() => new Tool.Failure({ message: "Use another tool" })),
|
|
|
- ),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ corrected: Tool.make({
|
|
|
+ description: "Fail with user correction feedback",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: () =>
|
|
|
+ Effect.fail(new PermissionV2.CorrectedError({ feedback: "Use another tool" })).pipe(
|
|
|
+ Effect.mapError(() => new Tool.Failure({ message: "Use another tool" })),
|
|
|
+ ),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* admit(session, "Call corrected")
|
|
|
|
|
|
responses = [reply.tool("call-corrected", "corrected", {}), reply.stop()]
|
|
|
@@ -3540,13 +3585,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
status: "error",
|
|
|
error: {
|
|
|
type: "unknown",
|
|
|
- message: expect.stringContaining("Failed to encode tool output"),
|
|
|
+ message: expect.stringContaining("Failed to write tool output"),
|
|
|
},
|
|
|
},
|
|
|
},
|
|
|
],
|
|
|
finish: "error",
|
|
|
- error: { type: "unknown", message: expect.stringContaining("Failed to encode tool output") },
|
|
|
+ error: { type: "unknown", message: expect.stringContaining("Failed to write tool output") },
|
|
|
},
|
|
|
])
|
|
|
}),
|
|
|
@@ -3594,14 +3639,17 @@ describe("SessionRunnerLLM", () => {
|
|
|
Effect.gen(function* () {
|
|
|
const session = yield* setup
|
|
|
const registry = yield* ToolRegistry.Service
|
|
|
- yield* registry.register({
|
|
|
- question: Tool.make({
|
|
|
- description: "Ask the user",
|
|
|
- input: Schema.Struct({}),
|
|
|
- output: Schema.Struct({}),
|
|
|
- execute: () => Effect.die(new QuestionTool.CancelledError()),
|
|
|
- }),
|
|
|
- }, { codemode: false })
|
|
|
+ yield* registry.register(
|
|
|
+ {
|
|
|
+ question: Tool.make({
|
|
|
+ description: "Ask the user",
|
|
|
+ input: Schema.Struct({}),
|
|
|
+ output: Schema.Struct({}),
|
|
|
+ execute: () => Effect.die(new QuestionTool.CancelledError()),
|
|
|
+ }),
|
|
|
+ },
|
|
|
+ { codemode: false },
|
|
|
+ )
|
|
|
yield* admit(session, "Ask then stop")
|
|
|
|
|
|
responses = [reply.tool("call-question", "question", {}), []]
|
|
|
@@ -3655,7 +3703,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
{
|
|
|
type: "assistant",
|
|
|
content: [
|
|
|
- { type: "tool", id: "call-before-failure", state: { status: "completed", structured: { text: "settle" } } },
|
|
|
+ {
|
|
|
+ type: "tool",
|
|
|
+ id: "call-before-failure",
|
|
|
+ state: { status: "completed", content: [{ type: "text", text: "settle" }] },
|
|
|
+ },
|
|
|
],
|
|
|
},
|
|
|
])
|
|
|
@@ -3663,7 +3715,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, assistant.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.success.1",
|
|
|
+ "session.tool.success.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
}),
|
|
|
@@ -3707,7 +3759,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, assistant.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
|
|
|
@@ -3808,7 +3860,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(requests).toHaveLength(2)
|
|
|
expect(requests[0]?.toolChoice).toBeUndefined()
|
|
|
expect(requests[1]?.toolChoice).toMatchObject({ type: "none" })
|
|
|
- expect(requests[1]?.tools).toEqual([])
|
|
|
+ // Protocols with native "none" keep these definitions for prompt caching.
|
|
|
+ expect(requests[1]?.tools.map((tool) => tool.name)).toContain("echo")
|
|
|
expect(requests[1]?.messages.at(-1)).toMatchObject({
|
|
|
role: "assistant",
|
|
|
content: [{ type: "text", text: expect.stringContaining("MAXIMUM STEPS REACHED") }],
|
|
|
@@ -3953,7 +4006,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(events.map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.success.1",
|
|
|
+ "session.tool.success.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
expect(
|
|
|
@@ -4146,7 +4199,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
content: [{ type: "text", text: expect.stringContaining("MAXIMUM STEPS REACHED") }],
|
|
|
})
|
|
|
expect(requests[2]?.toolChoice).toMatchObject({ type: "none" })
|
|
|
- expect(requests[2]?.tools).toEqual([])
|
|
|
+ // The final step keeps tool definitions to preserve provider prompt caching.
|
|
|
+ expect(requests[2]?.tools.map((tool) => tool.name)).toContain("echo")
|
|
|
expect(requests[2]?.messages.at(-1)).toMatchObject({
|
|
|
role: "assistant",
|
|
|
content: [{ type: "text", text: expect.stringContaining("MAXIMUM STEPS REACHED") }],
|
|
|
@@ -4197,7 +4251,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(yield* recordedStepSettlementEvents(sessionID, assistant.id)).toMatchObject([
|
|
|
{ type: "session.step.started.1" },
|
|
|
{
|
|
|
- type: "session.tool.failed.1",
|
|
|
+ type: "session.tool.failed.2",
|
|
|
data: {
|
|
|
callID: "call-malformed",
|
|
|
error: { type: "provider.invalid-output", message: "Invalid JSON input for tool call echo" },
|
|
|
@@ -4292,7 +4346,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(failed.error).toBeUndefined()
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, failed.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.ended.1",
|
|
|
])
|
|
|
const database = (yield* Database.Service).db
|
|
|
@@ -4521,7 +4575,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(requests).toHaveLength(2)
|
|
|
expect(requests[0]?.toolChoice).toBeUndefined()
|
|
|
expect(requests[1]?.toolChoice).toMatchObject({ type: "none" })
|
|
|
- expect((yield* recordedEventTypes(sessionID)).filter((type) => type === "session.tool.failed.1")).toHaveLength(2)
|
|
|
+ expect((yield* recordedEventTypes(sessionID)).filter((type) => type === "session.tool.failed.2")).toHaveLength(2)
|
|
|
}),
|
|
|
)
|
|
|
|
|
|
@@ -4553,7 +4607,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, assistant.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.success.1",
|
|
|
+ "session.tool.success.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
}),
|
|
|
@@ -4585,7 +4639,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect((yield* recordedStepSettlementEvents(sessionID, assistant.id)).map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
}),
|
|
|
@@ -4609,7 +4663,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(events.map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
expect(events[2]?.data.error).toMatchObject({ type: "unknown", message: "unexpected tool defect" })
|
|
|
@@ -4646,7 +4700,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(events.map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
expect(
|
|
|
@@ -4684,7 +4738,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(events.map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.ended.1",
|
|
|
])
|
|
|
expect(
|
|
|
@@ -4721,8 +4775,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
{ type: "session.step.started.1", callID: undefined },
|
|
|
{ type: "session.tool.called.1", callID: "call-local-raw-failure" },
|
|
|
{ type: "session.tool.called.1", callID: "call-hosted-raw-failure-pair" },
|
|
|
- { type: "session.tool.failed.1", callID: "call-local-raw-failure" },
|
|
|
- { type: "session.tool.failed.1", callID: "call-hosted-raw-failure-pair" },
|
|
|
+ { type: "session.tool.failed.2", callID: "call-local-raw-failure" },
|
|
|
+ { type: "session.tool.failed.2", callID: "call-hosted-raw-failure-pair" },
|
|
|
{ type: "session.step.failed.1", callID: undefined },
|
|
|
])
|
|
|
expect(
|
|
|
@@ -4748,7 +4802,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
expect(events.map((event) => event.type)).toEqual([
|
|
|
"session.step.started.1",
|
|
|
"session.tool.called.1",
|
|
|
- "session.tool.failed.1",
|
|
|
+ "session.tool.failed.2",
|
|
|
"session.step.failed.1",
|
|
|
])
|
|
|
expect(
|