|
|
|
@@ -1,8 +1,8 @@
|
|
|
|
|
import { describe, expect, test } from "bun:test"
|
|
|
|
|
import {
|
|
|
|
|
LLMClient,
|
|
|
|
|
LLMError,
|
|
|
|
|
LLMEvent,
|
|
|
|
|
LLMRequest,
|
|
|
|
|
Message,
|
|
|
|
|
Model,
|
|
|
|
|
SystemPart,
|
|
|
|
@@ -11,8 +11,6 @@ import {
|
|
|
|
|
InvalidProviderOutputReason,
|
|
|
|
|
InvalidRequestReason,
|
|
|
|
|
RateLimitReason,
|
|
|
|
|
type LLMClientShape,
|
|
|
|
|
type LLMRequest,
|
|
|
|
|
} from "@opencode-ai/ai"
|
|
|
|
|
import * as OpenAIChat from "@opencode-ai/ai/protocols/openai-chat"
|
|
|
|
|
import { Catalog } from "@opencode-ai/core/catalog"
|
|
|
|
@@ -74,81 +72,28 @@ import { Cause, DateTime, Deferred, Effect, Exit, Fiber, Layer, Schema, Scope, S
|
|
|
|
|
import { TestClock } from "effect/testing"
|
|
|
|
|
import { asc, eq } from "drizzle-orm"
|
|
|
|
|
import { testEffect } from "./lib/effect"
|
|
|
|
|
import { TestLLM } from "./lib/llm"
|
|
|
|
|
import { agentHost, catalogHost, host } from "./plugin/host"
|
|
|
|
|
import PROMPT_DEFAULT from "../src/session/runner/prompt/base.txt"
|
|
|
|
|
import { CodeModeInstructions } from "@opencode-ai/core/codemode/instructions"
|
|
|
|
|
|
|
|
|
|
const requests: LLMRequest[] = []
|
|
|
|
|
let requests: LLMRequest[] = []
|
|
|
|
|
const emptyCodeMode = `\n\n${CodeModeInstructions.render({ total: 0, shown: 0, namespaces: [] })}`
|
|
|
|
|
let response: LLMEvent[] = []
|
|
|
|
|
let responses: LLMEvent[][] | undefined
|
|
|
|
|
let responseStream: Stream.Stream<LLMEvent, LLMError> | undefined
|
|
|
|
|
let responseStreams: Stream.Stream<LLMEvent, LLMError>[] | undefined
|
|
|
|
|
let streamGate: Deferred.Deferred<void> | undefined
|
|
|
|
|
let streamStarted: Deferred.Deferred<void> | undefined
|
|
|
|
|
let streamFailure: LLMError | undefined
|
|
|
|
|
let toolExecutionGate: Deferred.Deferred<void> | undefined
|
|
|
|
|
let toolExecutionsStarted: Deferred.Deferred<void> | undefined
|
|
|
|
|
let toolExecutionsReady = 5
|
|
|
|
|
let activeToolExecutions = 0
|
|
|
|
|
let maxActiveToolExecutions = 0
|
|
|
|
|
const client = Layer.succeed(
|
|
|
|
|
LLMClient.Service,
|
|
|
|
|
LLMClient.Service.of({
|
|
|
|
|
prepare: () => Effect.die("unused"),
|
|
|
|
|
stream: ((request: LLMRequest) => {
|
|
|
|
|
requests.push({
|
|
|
|
|
...request,
|
|
|
|
|
system: request.system.map((part) => ({
|
|
|
|
|
...part,
|
|
|
|
|
text: part.text.replace(emptyCodeMode, ""),
|
|
|
|
|
})),
|
|
|
|
|
tools: request.tools.filter((tool) => tool.name !== "execute"),
|
|
|
|
|
})
|
|
|
|
|
if (responseStreams) return responseStreams.shift() ?? Stream.empty
|
|
|
|
|
if (responseStream) {
|
|
|
|
|
const stream = responseStream
|
|
|
|
|
responseStream = undefined
|
|
|
|
|
return stream
|
|
|
|
|
}
|
|
|
|
|
const bus = streamFailure
|
|
|
|
|
? Stream.fail(streamFailure)
|
|
|
|
|
: Stream.fromIterable(responses === undefined ? response : (responses.shift() ?? []))
|
|
|
|
|
if (!streamGate) return bus
|
|
|
|
|
return Stream.unwrap(
|
|
|
|
|
(streamStarted ? Deferred.succeed(streamStarted, undefined) : Effect.void).pipe(
|
|
|
|
|
Effect.andThen(Deferred.await(streamGate)),
|
|
|
|
|
Effect.as(bus),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
}) as unknown as LLMClientShape["stream"],
|
|
|
|
|
generate: () => Effect.die("unused"),
|
|
|
|
|
const testLLM = TestLLM.layer((request) =>
|
|
|
|
|
LLMRequest.update(request, {
|
|
|
|
|
system: request.system.map((part) => ({
|
|
|
|
|
...part,
|
|
|
|
|
text: part.text.replace(emptyCodeMode, ""),
|
|
|
|
|
})),
|
|
|
|
|
tools: request.tools.filter((tool) => tool.name !== "execute"),
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
const reply = {
|
|
|
|
|
stop: () => [
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
],
|
|
|
|
|
text: (text: string, id: string) => fragmentFixture("text", id, [text]).completeEvents,
|
|
|
|
|
textWithUsage: (text: string, id: string, inputTokens: number) =>
|
|
|
|
|
fragmentFixture("text", id, [text]).completeEvents.map((event) =>
|
|
|
|
|
LLMEvent.is.stepFinish(event)
|
|
|
|
|
? LLMEvent.stepFinish({
|
|
|
|
|
index: event.index,
|
|
|
|
|
reason: event.reason,
|
|
|
|
|
usage: { inputTokens, nonCachedInputTokens: inputTokens },
|
|
|
|
|
})
|
|
|
|
|
: event,
|
|
|
|
|
),
|
|
|
|
|
tool: (id: string, name: string, input: unknown) => [
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id, name, input }),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
],
|
|
|
|
|
}
|
|
|
|
|
const client = TestLLM.clientLayer
|
|
|
|
|
const model = Model.make({ id: "fake-model", provider: "fake", route: OpenAIChat.route })
|
|
|
|
|
const defaultSystem = PROMPT_DEFAULT
|
|
|
|
|
const replacementModel = Model.make({ id: "replacement", provider: "fake", route: OpenAIChat.route })
|
|
|
|
@@ -221,7 +166,7 @@ test("does not apply an ineligible tier without base pricing", () => {
|
|
|
|
|
|
|
|
|
|
const authorizations: Tool.Context[] = []
|
|
|
|
|
const executions: string[] = []
|
|
|
|
|
const permissionFail = ({
|
|
|
|
|
const permissionFail = {
|
|
|
|
|
name: "permission_fail",
|
|
|
|
|
description: "Reject a permission",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
@@ -235,7 +180,7 @@ const permissionFail = ({
|
|
|
|
|
resources: ["src/index.ts"],
|
|
|
|
|
}),
|
|
|
|
|
}),
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
const permission = Layer.succeed(
|
|
|
|
|
Permission.Service,
|
|
|
|
|
Permission.Service.of({
|
|
|
|
@@ -247,11 +192,7 @@ const permission = Layer.succeed(
|
|
|
|
|
list: () => Effect.die("unused"),
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
const transformTools = (
|
|
|
|
|
registry: Tool.Interface,
|
|
|
|
|
tools: Readonly<Record<string, Info>>,
|
|
|
|
|
options?: Tool.Options,
|
|
|
|
|
) =>
|
|
|
|
|
const transformTools = (registry: Tool.Interface, tools: Readonly<Record<string, Info>>, options?: Tool.Options) =>
|
|
|
|
|
registry.transform((draft) =>
|
|
|
|
|
Object.entries(tools).forEach(([name, tool]) =>
|
|
|
|
|
draft.add({ ...tool, name, options: { ...tool.options, ...options } }),
|
|
|
|
@@ -259,9 +200,10 @@ const transformTools = (
|
|
|
|
|
)
|
|
|
|
|
const echo = Layer.effectDiscard(
|
|
|
|
|
Tool.Service.use((registry) =>
|
|
|
|
|
transformTools(registry,
|
|
|
|
|
transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
echo: ({
|
|
|
|
|
echo: {
|
|
|
|
|
name: "echo",
|
|
|
|
|
description: "Echo text",
|
|
|
|
|
input: Schema.Struct({ text: Schema.String }),
|
|
|
|
@@ -278,8 +220,8 @@ const echo = Layer.effectDiscard(
|
|
|
|
|
if (toolExecutionGate) yield* Deferred.await(toolExecutionGate)
|
|
|
|
|
return { output: { text }, content: text }
|
|
|
|
|
}).pipe(Effect.ensuring(Effect.sync(() => activeToolExecutions--))),
|
|
|
|
|
}),
|
|
|
|
|
defect: ({
|
|
|
|
|
},
|
|
|
|
|
defect: {
|
|
|
|
|
name: "defect",
|
|
|
|
|
description: "Fail unexpectedly",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
@@ -288,14 +230,14 @@ const echo = Layer.effectDiscard(
|
|
|
|
|
(toolExecutionGate ? Deferred.await(toolExecutionGate) : Effect.void).pipe(
|
|
|
|
|
Effect.andThen(Effect.die("unexpected tool defect")),
|
|
|
|
|
),
|
|
|
|
|
}),
|
|
|
|
|
storefail: ({
|
|
|
|
|
},
|
|
|
|
|
storefail: {
|
|
|
|
|
name: "storefail",
|
|
|
|
|
description: "Produce output that cannot be persisted",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
|
output: Schema.Struct({}),
|
|
|
|
|
execute: () => Effect.succeed({ output: {} }),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
),
|
|
|
|
@@ -469,7 +411,7 @@ const it = testEffect(
|
|
|
|
|
[Config.node, config],
|
|
|
|
|
[PluginSupervisor.node, pluginSupervisor],
|
|
|
|
|
],
|
|
|
|
|
),
|
|
|
|
|
).pipe(Layer.provideMerge(testLLM)),
|
|
|
|
|
)
|
|
|
|
|
const sessionID = Session.ID.make("ses_runner_test")
|
|
|
|
|
const otherSessionID = Session.ID.make("ses_runner_other")
|
|
|
|
@@ -506,10 +448,9 @@ const setup = Effect.gen(function* () {
|
|
|
|
|
yield* Effect.forEach(SystemPromptPlugin.Plugins, (plugin) => plugin.effect(pluginHost), {
|
|
|
|
|
discard: true,
|
|
|
|
|
})
|
|
|
|
|
requests.length = 0
|
|
|
|
|
requests = (yield* TestLLM.Service).requests
|
|
|
|
|
authorizations.length = 0
|
|
|
|
|
executions.length = 0
|
|
|
|
|
response = []
|
|
|
|
|
systemBaseline = "Initial context"
|
|
|
|
|
systemRemoved = false
|
|
|
|
|
systemUnavailable = false
|
|
|
|
@@ -518,12 +459,6 @@ const setup = Effect.gen(function* () {
|
|
|
|
|
pluginFlushHook = Effect.void
|
|
|
|
|
currentModel = model
|
|
|
|
|
skillBaselines.clear()
|
|
|
|
|
responses = undefined
|
|
|
|
|
streamFailure = undefined
|
|
|
|
|
responseStream = undefined
|
|
|
|
|
responseStreams = undefined
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
toolExecutionGate = undefined
|
|
|
|
|
toolExecutionsStarted = undefined
|
|
|
|
|
toolExecutionsReady = 5
|
|
|
|
@@ -567,7 +502,7 @@ const rateLimited = (retryAfterMs?: number) =>
|
|
|
|
|
|
|
|
|
|
const setupOverflowRecovery = Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.text("Earlier answer", "text-earlier")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Earlier answer", "text-earlier"))
|
|
|
|
|
yield* admit(session, "Earlier question ".repeat(700))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
currentModel = recoveryModel
|
|
|
|
@@ -742,7 +677,7 @@ const verifyEphemeralDeltas = (kind: FragmentKind) =>
|
|
|
|
|
const bus = yield* Bus.Service
|
|
|
|
|
const live = yield* bus.subscribe(fixture.delta).pipe(Stream.take(32), Stream.runCollect, Effect.forkScoped)
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
response = fixture.completeEvents
|
|
|
|
|
yield* TestLLM.push(fixture.completeEvents)
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -769,7 +704,7 @@ const verifyPartialFlushOnFailure = (kind: FragmentKind) =>
|
|
|
|
|
const fixture = fragmentFixture(kind, fragmentID(kind, "partial"), ["Partial"])
|
|
|
|
|
const failure = providerUnavailable()
|
|
|
|
|
yield* admit(session, prompt)
|
|
|
|
|
responseStream = Stream.concat(Stream.fromIterable(fixture.partialEvents), Stream.fail(failure))
|
|
|
|
|
yield* TestLLM.push(Stream.concat(Stream.fromIterable(fixture.partialEvents), Stream.fail(failure)))
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
expect(yield* session.context(sessionID)).toMatchObject([
|
|
|
|
@@ -802,9 +737,11 @@ const verifyPartialFlushOnInterruption = (kind: FragmentKind) =>
|
|
|
|
|
const fixture = fragmentFixture(kind, fragmentID(kind, "interrupted"), ["Partial"])
|
|
|
|
|
const streamed = yield* Deferred.make<void>()
|
|
|
|
|
yield* admit(session, prompt)
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
Stream.fromIterable(fixture.partialEvents),
|
|
|
|
|
Stream.fromEffect(Deferred.succeed(streamed, undefined)).pipe(Stream.flatMap(() => Stream.never)),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(
|
|
|
|
|
Stream.fromIterable(fixture.partialEvents),
|
|
|
|
|
Stream.fromEffect(Deferred.succeed(streamed, undefined)).pipe(Stream.flatMap(() => Stream.never)),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const runner = yield* SessionRunner.Service
|
|
|
|
@@ -840,7 +777,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Original message")
|
|
|
|
|
responses = [reply.tool("call-removed", "echo", { text: "blocked" })]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-removed", "echo", { text: "blocked" }))
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -872,9 +809,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
const contexts: Tool.Context[] = []
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
location_context: ({
|
|
|
|
|
location_context: {
|
|
|
|
|
name: "location_context",
|
|
|
|
|
description: "Read application context",
|
|
|
|
|
input: Schema.Struct({ query: Schema.String }),
|
|
|
|
@@ -885,12 +823,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* context.progress({ phase: "reading" })
|
|
|
|
|
return { output: { answer: query.toUpperCase() } }
|
|
|
|
|
}),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Use application context")
|
|
|
|
|
responses = [reply.tool("call-location", "location_context", { query: "hello" }), []]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-location", "location_context", { query: "hello" }), [])
|
|
|
|
|
const bus = yield* Bus.Service
|
|
|
|
|
const progressFiber = yield* bus.subscribe(SessionEvent.Tool.Progress).pipe(
|
|
|
|
|
Stream.filter((event) => event.data.sessionID === sessionID && event.data.callID === "call-location"),
|
|
|
|
@@ -934,22 +872,22 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
const scope = yield* Scope.make()
|
|
|
|
|
const executions: string[] = []
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
{
|
|
|
|
|
reloaded: ({
|
|
|
|
|
name: "reloaded",
|
|
|
|
|
description: "Record the advertised tool",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
|
output: Schema.Struct({ value: Schema.String }),
|
|
|
|
|
execute: () =>
|
|
|
|
|
Effect.sync(() => executions.push("advertised")).pipe(Effect.as({ output: { value: "advertised" } })),
|
|
|
|
|
}),
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
reloaded: {
|
|
|
|
|
name: "reloaded",
|
|
|
|
|
description: "Record the advertised tool",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
|
output: Schema.Struct({ value: Schema.String }),
|
|
|
|
|
execute: () =>
|
|
|
|
|
Effect.sync(() => executions.push("advertised")).pipe(Effect.as({ output: { value: "advertised" } })),
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
.pipe(Scope.provide(scope))
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
).pipe(Scope.provide(scope))
|
|
|
|
|
yield* admit(session, "Use the reloaded tool")
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-reloaded", name: "reloaded", input: {} }),
|
|
|
|
@@ -957,27 +895,27 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
],
|
|
|
|
|
[],
|
|
|
|
|
]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
)
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* Scope.close(scope, Exit.void)
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
reloaded: ({
|
|
|
|
|
reloaded: {
|
|
|
|
|
name: "reloaded",
|
|
|
|
|
description: "Record the replacement tool",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
|
output: Schema.Struct({ value: Schema.String }),
|
|
|
|
|
execute: () =>
|
|
|
|
|
Effect.sync(() => executions.push("replacement")).pipe(Effect.as({ output: { value: "replacement" } })),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(run)
|
|
|
|
|
|
|
|
|
|
expect(executions).toEqual(["advertised"])
|
|
|
|
@@ -1019,16 +957,16 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const secondStarted = yield* Deferred.make<void>()
|
|
|
|
|
const releaseSecond = yield* Deferred.make<void>()
|
|
|
|
|
responseStreams = [
|
|
|
|
|
Stream.fromIterable(reply.tool("call-echo", "echo", { text: "background started" })),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.fromIterable(TestLLM.tool("call-echo", "echo", { text: "background started" })),
|
|
|
|
|
Stream.unwrap(
|
|
|
|
|
Deferred.succeed(secondStarted, undefined).pipe(
|
|
|
|
|
Effect.andThen(Deferred.await(releaseSecond)),
|
|
|
|
|
Effect.as(Stream.fromIterable(reply.stop())),
|
|
|
|
|
Effect.as(Stream.fromIterable(TestLLM.stop())),
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
Stream.fromIterable(reply.text("Handled completion", "text-completion")),
|
|
|
|
|
]
|
|
|
|
|
Stream.fromIterable(TestLLM.text("Handled completion", "text-completion")),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Start background work")
|
|
|
|
|
const running = yield* session.resume(sessionID).pipe(Effect.forkChild({ startImmediately: true }))
|
|
|
|
|
yield* Deferred.await(secondStarted)
|
|
|
|
@@ -1038,7 +976,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* Fiber.join(running)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(3)
|
|
|
|
|
expect(userTexts(requests[2]!)).toContain("Background work completed")
|
|
|
|
|
expect(userTexts(requests[2])).toContain("Background work completed")
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -1145,7 +1083,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("forks instruction values at the selected message instead of the parent's latest state", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const first = yield* admit(session, "First")
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
systemBaseline = "Changed context"
|
|
|
|
|
const second = yield* admit(session, "Second")
|
|
|
|
@@ -1224,6 +1162,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
initial_values: { "test/context": Instructions.hash("Initial context") },
|
|
|
|
|
current_values: { "test/context": Instructions.hash("Initial context") },
|
|
|
|
|
})
|
|
|
|
|
return undefined
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -1268,8 +1207,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
|
|
|
|
|
expect(
|
|
|
|
|
PromptCacheDiagnostics.compare(
|
|
|
|
|
PromptCacheDiagnostics.snapshot(requests[0]!),
|
|
|
|
|
PromptCacheDiagnostics.snapshot(requests[1]!),
|
|
|
|
|
PromptCacheDiagnostics.snapshot(requests[0]),
|
|
|
|
|
PromptCacheDiagnostics.snapshot(requests[1]),
|
|
|
|
|
),
|
|
|
|
|
).toEqual({ status: "append-only", previousMessages: 1, currentMessages: 3 })
|
|
|
|
|
expect(requests.map((request) => request.system.map((part) => part.text))).toEqual([
|
|
|
|
@@ -1307,7 +1246,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
currentModel = Model.make({ id: "gpt-5", provider: "openai", route: OpenAIChat.route })
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Done", "text-provider-prompt")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Done", "text-provider-prompt"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests.at(-1)?.system.map((part) => part.text)).toEqual([
|
|
|
|
@@ -1330,7 +1269,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Done", "text-empty-agent-system")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Done", "text-empty-agent-system"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests.at(-1)?.system.map((part) => part.text)).toEqual([
|
|
|
|
@@ -1352,7 +1291,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Done", "text-build")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Done", "text-build"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests.at(-1)?.system.map((part) => part.text)).toEqual(["Build agent instructions", "Initial context"])
|
|
|
|
@@ -1376,7 +1315,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
})
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Done", "text-reviewer")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Done", "text-reviewer"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests.at(-1)?.system.map((part) => part.text)).toEqual(["Reviewer instructions", "Initial context"])
|
|
|
|
@@ -1396,7 +1335,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Done", "text-no-system")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Done", "text-no-system"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests.at(-1)?.system.map((part) => part.text)).toEqual(["Build agent instructions", "Initial context"])
|
|
|
|
@@ -1422,7 +1361,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
.pipe(Effect.orDie)
|
|
|
|
|
yield* admit(session, "First")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Done", "text-selected")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Done", "text-selected"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests.at(-1)?.system.map((part) => part.text)).toEqual(["Reviewer instructions", "Initial context"])
|
|
|
|
@@ -1444,7 +1383,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Inspect files", resume: false })
|
|
|
|
|
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
const failure = yield* session.resume(sessionID).pipe(Effect.flip)
|
|
|
|
|
|
|
|
|
|
expect(failure).toMatchObject({
|
|
|
|
@@ -1465,7 +1404,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Wait for plugins", resume: false })
|
|
|
|
|
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
const running = yield* session.resume(sessionID).pipe(Effect.forkChild({ startImmediately: true }))
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
|
|
|
|
@@ -1504,7 +1443,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
[defaultSystem, "Initial context\n\nBuild skills"],
|
|
|
|
|
[defaultSystem, "Initial context\n\nBuild skills"],
|
|
|
|
|
])
|
|
|
|
|
expect(systemTexts(requests[1]!)).toContainEqual(expect.stringContaining("Reviewer skills"))
|
|
|
|
|
expect(systemTexts(requests[1])).toContainEqual(expect.stringContaining("Reviewer skills"))
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -1772,17 +1711,16 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
currentModel = recoveryModel
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
responses = [
|
|
|
|
|
reply.tool("call-active", "echo", { text: "active" }),
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.tool("call-active", "echo", { text: "active" }),
|
|
|
|
|
[LLMEvent.textDelta({ id: "summary", text: "durable summary" })],
|
|
|
|
|
reply.text("Steer complete", "text-steer"),
|
|
|
|
|
reply.text("Queue complete", "text-queue"),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.text("Steer complete", "text-steer"),
|
|
|
|
|
TestLLM.text("Queue complete", "text-queue"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Active work")
|
|
|
|
|
const active = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
|
|
|
|
|
const first = yield* session.compact({ sessionID })
|
|
|
|
|
const second = yield* session.compact({ sessionID })
|
|
|
|
@@ -1802,7 +1740,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
})
|
|
|
|
|
expect(yield* SessionPending.has((yield* Database.Service).db, sessionID, "steer")).toBe(false)
|
|
|
|
|
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(active)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(4)
|
|
|
|
@@ -1823,16 +1761,15 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
currentModel = recoveryModel
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
responses = [
|
|
|
|
|
reply.text("Active complete", "text-active-failure"),
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.text("Active complete", "text-active-failure"),
|
|
|
|
|
[],
|
|
|
|
|
reply.text("Continued", "text-after-failure"),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.text("Continued", "text-after-failure"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Active work")
|
|
|
|
|
const active = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
|
|
|
|
|
const compaction = yield* session.compact({ sessionID })
|
|
|
|
|
yield* session.prompt({
|
|
|
|
@@ -1841,7 +1778,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
delivery: "queue",
|
|
|
|
|
resume: false,
|
|
|
|
|
})
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(active)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(3)
|
|
|
|
@@ -1886,12 +1823,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("manually compacts when the model has no context limit", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.text("Earlier answer", "text-manual-unknown-history")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Earlier answer", "text-manual-unknown-history"))
|
|
|
|
|
yield* admit(session, "Earlier question")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = reply.text("Manual summary", "text-manual-unknown-summary")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Manual summary", "text-manual-unknown-summary"))
|
|
|
|
|
const compaction = yield* session.compact({ sessionID })
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -1908,11 +1845,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("preserves provider errors from manual compaction", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.text("Earlier answer", "text-manual-provider-history")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Earlier answer", "text-manual-provider-history"))
|
|
|
|
|
yield* admit(session, "Earlier question")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
response = [LLMEvent.providerError({ message: "summary unavailable" })]
|
|
|
|
|
yield* TestLLM.push([LLMEvent.providerError({ message: "summary unavailable" })])
|
|
|
|
|
const compaction = yield* session.compact({ sessionID })
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -1927,11 +1864,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("preserves typed provider failures from manual compaction", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.text("Earlier answer", "text-manual-failure-history")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Earlier answer", "text-manual-failure-history"))
|
|
|
|
|
yield* admit(session, "Earlier question")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
responseStream = Stream.fail(providerUnavailable())
|
|
|
|
|
yield* TestLLM.push(Stream.fail(providerUnavailable()))
|
|
|
|
|
const compaction = yield* session.compact({ sessionID })
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -1946,15 +1883,17 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("records cancelled manual compaction without surfacing an internal failure", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.text("Earlier answer", "text-manual-interrupt-history")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Earlier answer", "text-manual-interrupt-history"))
|
|
|
|
|
yield* admit(session, "Earlier question")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
const streamed = yield* Deferred.make<void>()
|
|
|
|
|
const partial = fragmentFixture("text", "text-manual-interrupt-summary", ["Partial summary"])
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
Stream.fromIterable(partial.partialEvents),
|
|
|
|
|
Stream.fromEffect(Deferred.succeed(streamed, undefined)).pipe(Stream.flatMap(() => Stream.never)),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(
|
|
|
|
|
Stream.fromIterable(partial.partialEvents),
|
|
|
|
|
Stream.fromEffect(Deferred.succeed(streamed, undefined)).pipe(Stream.flatMap(() => Stream.never)),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
const compaction = yield* session.compact({ sessionID })
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
@@ -1975,7 +1914,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("settles an admitted manual compaction when pre-start resolution throws", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.text("Earlier answer", "text-manual-resolution-history")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Earlier answer", "text-manual-resolution-history"))
|
|
|
|
|
yield* admit(session, "Earlier question")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2001,16 +1940,16 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("automatically compacts into a completed summary and retained recent turn", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.textWithUsage("Earlier answer", "text-first", 3_950)
|
|
|
|
|
yield* TestLLM.push(TestLLM.textWithUsage("Earlier answer", "text-first", 3_950))
|
|
|
|
|
yield* admit(session, "Earlier question ".repeat(180))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
currentModel = compactModel
|
|
|
|
|
requests.length = 0
|
|
|
|
|
responses = [
|
|
|
|
|
reply.text("## Objective\n- Preserve the task", "text-summary"),
|
|
|
|
|
reply.textWithUsage("Continued", "text-final", 3_950),
|
|
|
|
|
]
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.text("## Objective\n- Preserve the task", "text-summary"),
|
|
|
|
|
TestLLM.textWithUsage("Continued", "text-final", 3_950),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Recent exact request ".repeat(180))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2029,10 +1968,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
|
|
|
|
|
requests.length = 0
|
|
|
|
|
executions.length = 0
|
|
|
|
|
responses = [
|
|
|
|
|
reply.text("## Objective\n- Preserve the updated task", "text-summary-2"),
|
|
|
|
|
reply.text("Continued again", "text-final-2"),
|
|
|
|
|
]
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.text("## Objective\n- Preserve the updated task", "text-summary-2"),
|
|
|
|
|
TestLLM.text("Continued again", "text-final-2"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Newest exact request ".repeat(180))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2052,12 +1991,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
currentModel = fullOutputModel
|
|
|
|
|
response = reply.textWithUsage("Earlier answer", "text-full-output-first", 9_500)
|
|
|
|
|
yield* TestLLM.push(TestLLM.textWithUsage("Earlier answer", "text-full-output-first", 9_500))
|
|
|
|
|
yield* admit(session, "Earlier question")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = reply.text("Continued", "text-full-output-final")
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Continued", "text-full-output-final"))
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2070,16 +2009,16 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("stops after required automatic compaction fails", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = reply.textWithUsage("Earlier answer", "text-before-failed-compaction", 3_950)
|
|
|
|
|
yield* TestLLM.push(TestLLM.textWithUsage("Earlier answer", "text-before-failed-compaction", 3_950))
|
|
|
|
|
yield* admit(session, "Earlier question ".repeat(180))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
currentModel = compactModel
|
|
|
|
|
requests.length = 0
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[LLMEvent.providerError({ message: "Unsupported parameter: max_output_tokens" })],
|
|
|
|
|
reply.text("Must not run", "text-after-failed-compaction"),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.text("Must not run", "text-after-failed-compaction"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Recent exact request ".repeat(180))
|
|
|
|
|
expect(yield* Effect.exit(session.resume(sessionID))).toMatchObject({ _tag: "Failure" })
|
|
|
|
|
|
|
|
|
@@ -2099,14 +2038,14 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("forces one compaction and retries after provider context overflow", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setupOverflowRecovery
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" }),
|
|
|
|
|
],
|
|
|
|
|
reply.text("## Objective\n- Recover overflow", "text-summary"),
|
|
|
|
|
reply.text("Recovered", "text-final"),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.text("## Objective\n- Recover overflow", "text-summary"),
|
|
|
|
|
TestLLM.text("Recovered", "text-final"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2129,11 +2068,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setupOverflowRecovery
|
|
|
|
|
currentModel = model
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" })],
|
|
|
|
|
reply.text("## Objective\n- Recover unknown limit", "text-summary-unknown-limit"),
|
|
|
|
|
reply.text("Recovered", "text-final-unknown-limit"),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.text("## Objective\n- Recover unknown limit", "text-summary-unknown-limit"),
|
|
|
|
|
TestLLM.text("Recovered", "text-final-unknown-limit"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2149,11 +2088,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setupOverflowRecovery
|
|
|
|
|
currentModel = undersizedContextModel
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" })],
|
|
|
|
|
reply.text("## Objective\n- Recover undersized limit", "text-summary-undersized-limit"),
|
|
|
|
|
reply.text("Recovered", "text-final-undersized-limit"),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.text("## Objective\n- Recover undersized limit", "text-summary-undersized-limit"),
|
|
|
|
|
TestLLM.text("Recovered", "text-final-undersized-limit"),
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2172,7 +2111,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" }),
|
|
|
|
|
]
|
|
|
|
|
responses = [overflow(), reply.text("## Objective\n- Recover once", "text-summary"), overflow()]
|
|
|
|
|
yield* TestLLM.push(overflow(), TestLLM.text("## Objective\n- Recover once", "text-summary"), overflow())
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("prompt too long")
|
|
|
|
|
|
|
|
|
@@ -2187,20 +2126,22 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("recovers once from a raw context overflow failure", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setupOverflowRecovery
|
|
|
|
|
responseStream = Stream.fail(
|
|
|
|
|
new LLMError({
|
|
|
|
|
module: "test",
|
|
|
|
|
method: "stream",
|
|
|
|
|
reason: new InvalidRequestReason({
|
|
|
|
|
message: "prompt too long",
|
|
|
|
|
classification: "context-overflow",
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.fail(
|
|
|
|
|
new LLMError({
|
|
|
|
|
module: "test",
|
|
|
|
|
method: "stream",
|
|
|
|
|
reason: new InvalidRequestReason({
|
|
|
|
|
message: "prompt too long",
|
|
|
|
|
classification: "context-overflow",
|
|
|
|
|
}),
|
|
|
|
|
}),
|
|
|
|
|
}),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.text("## Objective\n- Recover raw overflow", "text-summary"),
|
|
|
|
|
TestLLM.text("Recovered", "text-final"),
|
|
|
|
|
)
|
|
|
|
|
responses = [
|
|
|
|
|
reply.text("## Objective\n- Recover raw overflow", "text-summary"),
|
|
|
|
|
reply.text("Recovered", "text-final"),
|
|
|
|
|
]
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2215,10 +2156,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("publishes the original overflow when recovery summarization fails", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setupOverflowRecovery
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" })],
|
|
|
|
|
[LLMEvent.providerError({ message: "summary unavailable" })],
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("prompt too long")
|
|
|
|
|
|
|
|
|
@@ -2243,26 +2184,29 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("interrupts overflow recovery while the summary provider is running", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setupOverflowRecovery
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" })],
|
|
|
|
|
reply.text("## Objective\n- Interrupted", "text-summary"),
|
|
|
|
|
]
|
|
|
|
|
const firstGate = yield* Deferred.make<void>()
|
|
|
|
|
const summaryGate = yield* Deferred.make<void>()
|
|
|
|
|
streamGate = firstGate
|
|
|
|
|
TestLLM.text("## Objective\n- Interrupted", "text-summary"),
|
|
|
|
|
)
|
|
|
|
|
const first = yield* TestLLM.gate
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 1) yield* Effect.yieldNow
|
|
|
|
|
streamGate = summaryGate
|
|
|
|
|
yield* Deferred.succeed(firstGate, undefined)
|
|
|
|
|
while (requests.length < 2) yield* Effect.yieldNow
|
|
|
|
|
yield* first.started
|
|
|
|
|
|
|
|
|
|
const summary = yield* TestLLM.gate
|
|
|
|
|
yield* first.release
|
|
|
|
|
yield* summary.started
|
|
|
|
|
|
|
|
|
|
yield* session.interrupt(sessionID)
|
|
|
|
|
expect(yield* Fiber.await(run)).toMatchObject({ _tag: "Failure" })
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
const exit = yield* Fiber.await(run)
|
|
|
|
|
expect(Exit.isFailure(exit) && Cause.hasInterruptsOnly(exit.cause)).toBeTrue()
|
|
|
|
|
expect(yield* session.context(sessionID)).toContainEqual(
|
|
|
|
|
expect.objectContaining({ type: "compaction", status: "failed", reason: "auto" }),
|
|
|
|
|
expect.objectContaining({
|
|
|
|
|
type: "compaction",
|
|
|
|
|
status: "failed",
|
|
|
|
|
reason: "auto",
|
|
|
|
|
error: { type: "compaction.interrupted", message: "Compaction was interrupted" },
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
@@ -2303,7 +2247,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Use tools")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.reasoningStart({ id: "reasoning-1" }),
|
|
|
|
|
LLMEvent.reasoningDelta({ id: "reasoning-1", text: "Think" }),
|
|
|
|
@@ -2346,7 +2290,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
},
|
|
|
|
|
}),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2398,7 +2342,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Echo this")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-echo", "echo", { text: "hello" }), reply.text("Done", "text-final")]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-echo", "echo", { text: "hello" }), TestLLM.text("Done", "text-final"))
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2443,7 +2387,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const bus = yield* Bus.Service
|
|
|
|
|
yield* admit(session, "Echo this")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-echo", "echo", { text: "hello" }), reply.stop()]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-echo", "echo", { text: "hello" }), TestLLM.stop())
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsStarted = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsReady = 1
|
|
|
|
@@ -2462,7 +2406,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
[defaultSystem, "Initial context"],
|
|
|
|
|
[defaultSystem, "Initial context"],
|
|
|
|
|
])
|
|
|
|
|
expect(systemTexts(requests[1]!)).toContain("Replacement context")
|
|
|
|
|
expect(systemTexts(requests[1])).toContain("Replacement context")
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -2471,7 +2415,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Think first")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.reasoningStart({ id: "reasoning-anthropic" }),
|
|
|
|
|
LLMEvent.reasoningDelta({ id: "reasoning-anthropic", text: "Signed thought" }),
|
|
|
|
@@ -2496,7 +2440,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
}),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
yield* replaySessionProjection(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2520,7 +2464,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests[1]?.messages[1]?.content).toEqual([
|
|
|
|
@@ -2543,7 +2487,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Check first")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "commentary", providerMetadata: { openai: { phase: "commentary" } } }),
|
|
|
|
|
LLMEvent.textDelta({ id: "commentary", text: "Checking." }),
|
|
|
|
@@ -2553,7 +2497,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
}),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
yield* replaySessionProjection(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2566,7 +2510,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests[1]?.messages[1]?.content).toEqual([
|
|
|
|
@@ -2584,7 +2528,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Search first")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({
|
|
|
|
|
id: "hosted-search",
|
|
|
|
@@ -2602,12 +2546,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
}),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
yield* replaySessionProjection(sessionID)
|
|
|
|
|
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests[1]?.messages.map((message) => message.role)).toEqual(["user", "assistant", "user"])
|
|
|
|
@@ -2651,9 +2595,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
])
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
initial,
|
|
|
|
|
Stream.fromEffect(Deferred.await(providerGate)).pipe(Stream.flatMap(() => final)),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(initial, Stream.fromEffect(Deferred.await(providerGate)).pipe(Stream.flatMap(() => final))),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
@@ -2693,11 +2636,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Echo twice")
|
|
|
|
|
|
|
|
|
|
responses = [
|
|
|
|
|
reply.tool("tool_0", "echo", { text: "first" }),
|
|
|
|
|
reply.tool("tool_0", "echo", { text: "second" }),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.tool("tool_0", "echo", { text: "first" }),
|
|
|
|
|
TestLLM.tool("tool_0", "echo", { text: "second" }),
|
|
|
|
|
[],
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -2760,21 +2703,18 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Run once")
|
|
|
|
|
|
|
|
|
|
response = reply.text("Once", "text-once")
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Once", "text-once"))
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
const second = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
yield* Fiber.join(second)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
expect(yield* session.context(sessionID)).toMatchObject([
|
|
|
|
@@ -2789,22 +2729,19 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Start working")
|
|
|
|
|
|
|
|
|
|
responses = [reply.stop(), reply.stop()]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop(), TestLLM.stop())
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Change direction" })
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start working", "Change direction"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start working", "Change direction"])
|
|
|
|
|
expect((yield* session.context(sessionID)).map((message) => message.type)).toEqual([
|
|
|
|
|
"user",
|
|
|
|
|
"assistant",
|
|
|
|
@@ -2819,26 +2756,23 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Start working")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-echo", "echo", { text: "hello" }), reply.stop(), reply.stop()]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-echo", "echo", { text: "hello" }), TestLLM.stop(), TestLLM.stop())
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({
|
|
|
|
|
sessionID,
|
|
|
|
|
text: "Wait until continuation ends",
|
|
|
|
|
delivery: "queue",
|
|
|
|
|
})
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(3)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[2]!)).toEqual(["Start working", "Wait until continuation ends"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[2])).toEqual(["Start working", "Wait until continuation ends"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -2848,12 +2782,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const { db } = yield* Database.Service
|
|
|
|
|
yield* admit(session, "Interrupt current work")
|
|
|
|
|
|
|
|
|
|
responses = [[], reply.stop()]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push([], TestLLM.stop())
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({
|
|
|
|
|
sessionID,
|
|
|
|
|
text: "Run after interrupt",
|
|
|
|
@@ -2864,15 +2797,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
expect(yield* SessionPending.has(db, sessionID, "queue")).toBe(true)
|
|
|
|
|
const resumed = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 2) yield* Effect.yieldNow
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(resumed)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Interrupt current work"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Interrupt current work", "Run after interrupt"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Interrupt current work"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Interrupt current work", "Run after interrupt"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -2882,12 +2813,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const { db } = yield* Database.Service
|
|
|
|
|
yield* admit(session, "Interrupt current work")
|
|
|
|
|
|
|
|
|
|
responses = [[], reply.stop()]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push([], TestLLM.stop())
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({
|
|
|
|
|
sessionID,
|
|
|
|
|
text: "Steer after interrupt",
|
|
|
|
@@ -2898,15 +2828,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
expect(yield* SessionPending.has(db, sessionID, "steer")).toBe(true)
|
|
|
|
|
|
|
|
|
|
const resumed = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 2) yield* Effect.yieldNow
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(resumed)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Interrupt current work"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Interrupt current work", "Steer after interrupt"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Interrupt current work"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Interrupt current work", "Steer after interrupt"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -2915,23 +2843,20 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Start working")
|
|
|
|
|
|
|
|
|
|
responses = [reply.stop(), reply.stop(), reply.stop()]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop(), TestLLM.stop(), TestLLM.stop())
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Queue first", delivery: "queue" })
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Queue second", delivery: "queue" })
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(3)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start working", "Queue first"])
|
|
|
|
|
expect(userTexts(requests[2]!)).toEqual(["Start working", "Queue first", "Queue second"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start working", "Queue first"])
|
|
|
|
|
expect(userTexts(requests[2])).toEqual(["Start working", "Queue first", "Queue second"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -2946,13 +2871,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
resume: false,
|
|
|
|
|
})
|
|
|
|
|
|
|
|
|
|
responses = [reply.stop(), reply.stop()]
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop(), TestLLM.stop())
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Start steering"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start steering", "Queue for later"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Start steering"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start steering", "Queue for later"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -2961,39 +2886,36 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Start working")
|
|
|
|
|
|
|
|
|
|
responses = [reply.stop(), reply.stop(), reply.stop(), reply.stop()]
|
|
|
|
|
const firstGate = yield* Deferred.make<void>()
|
|
|
|
|
const secondGate = yield* Deferred.make<void>()
|
|
|
|
|
streamGate = firstGate
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop(), TestLLM.stop(), TestLLM.stop(), TestLLM.stop())
|
|
|
|
|
const firstStream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 1) yield* Effect.yieldNow
|
|
|
|
|
yield* firstStream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Queue first", delivery: "queue" })
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Queue second", delivery: "queue" })
|
|
|
|
|
streamGate = secondGate
|
|
|
|
|
yield* Deferred.succeed(firstGate, undefined)
|
|
|
|
|
while (requests.length < 2) yield* Effect.yieldNow
|
|
|
|
|
const secondStream = yield* TestLLM.gate
|
|
|
|
|
yield* firstStream.release
|
|
|
|
|
yield* secondStream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Steer before next queued input" })
|
|
|
|
|
yield* session.prompt({
|
|
|
|
|
sessionID,
|
|
|
|
|
text: "Also steer before next queued input",
|
|
|
|
|
})
|
|
|
|
|
yield* session.synthetic({ sessionID, text: "Background completion before next queued input" })
|
|
|
|
|
yield* Deferred.succeed(secondGate, undefined)
|
|
|
|
|
yield* secondStream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(4)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start working", "Queue first"])
|
|
|
|
|
expect(userTexts(requests[2]!)).toEqual([
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Start working"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start working", "Queue first"])
|
|
|
|
|
expect(userTexts(requests[2])).toEqual([
|
|
|
|
|
"Start working",
|
|
|
|
|
"Queue first",
|
|
|
|
|
"Steer before next queued input",
|
|
|
|
|
"Also steer before next queued input",
|
|
|
|
|
"Background completion before next queued input",
|
|
|
|
|
])
|
|
|
|
|
expect(userTexts(requests[3]!)).toEqual([
|
|
|
|
|
expect(userTexts(requests[3])).toEqual([
|
|
|
|
|
"Start working",
|
|
|
|
|
"Queue first",
|
|
|
|
|
"Steer before next queued input",
|
|
|
|
@@ -3009,22 +2931,19 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Start working")
|
|
|
|
|
|
|
|
|
|
responses = [reply.stop(), reply.stop()]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop(), TestLLM.stop())
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "First steer" })
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Second steer" })
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start working", "First steer", "Second steer"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start working", "First steer", "Second steer"])
|
|
|
|
|
yield* (yield* SessionExecution.Service).wake(sessionID)
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
@@ -3036,23 +2955,21 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Start working")
|
|
|
|
|
|
|
|
|
|
streamFailure = invalidRequest()
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
const failure = invalidRequest()
|
|
|
|
|
yield* TestLLM.push(Stream.fail(failure))
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Recover with this" })
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
expect(yield* Fiber.join(first).pipe(Effect.flip)).toBe(streamFailure)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
expect(yield* Fiber.join(first).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
|
|
|
|
|
streamFailure = undefined
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.wait(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(userTexts(requests[1]!)).toEqual(["Start working", "Recover with this"])
|
|
|
|
|
expect(userTexts(requests[1])).toEqual(["Start working", "Recover with this"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -3089,7 +3006,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
executed: false,
|
|
|
|
|
})
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -3147,7 +3064,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
state: { itemId: "call-hosted-interrupted" },
|
|
|
|
|
})
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -3184,7 +3101,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
name: "echo",
|
|
|
|
|
})
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -3206,11 +3123,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
resume: false,
|
|
|
|
|
})
|
|
|
|
|
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
yield* (yield* SessionExecution.Service).wake(sessionID)
|
|
|
|
|
while (requests.length === 0) yield* Effect.yieldNow
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* stream.release
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Wait in queue"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Wait in queue"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -3226,12 +3145,14 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.catchDefect(Effect.succeed))).toBe(defect)
|
|
|
|
|
fail = false
|
|
|
|
|
requests.length = 0
|
|
|
|
|
response = reply.stop()
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop())
|
|
|
|
|
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
yield* (yield* SessionExecution.Service).wake(sessionID)
|
|
|
|
|
while (requests.length === 0) yield* Effect.yieldNow
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* stream.release
|
|
|
|
|
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Recover promoted input"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Recover promoted input"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -3249,7 +3170,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
expect(userTexts(requests[0]!)).toEqual(["Run committed promotion"])
|
|
|
|
|
expect(userTexts(requests[0])).toEqual(["Run committed promotion"])
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -3301,25 +3222,21 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
resume: false,
|
|
|
|
|
})
|
|
|
|
|
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* stream.started
|
|
|
|
|
const second = yield* session.resume(otherSessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
expect(requests.map((request) => request.providerOptions?.openai?.promptCacheKey)).toEqual([
|
|
|
|
|
sessionID,
|
|
|
|
|
otherSessionID,
|
|
|
|
|
])
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(first)
|
|
|
|
|
yield* Fiber.join(second)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
}),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
@@ -3356,23 +3273,20 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Retry after failure")
|
|
|
|
|
|
|
|
|
|
streamFailure = invalidRequest()
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(Stream.fail(invalidRequest()))
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const first = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
const second = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Effect.yieldNow
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
const [firstExit, secondExit] = yield* Effect.all([Fiber.await(first), Fiber.await(second)])
|
|
|
|
|
expect(secondExit).toEqual(firstExit)
|
|
|
|
|
|
|
|
|
|
streamFailure = undefined
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
|
}),
|
|
|
|
@@ -3383,7 +3297,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Call missing")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-missing", "missing", {}), reply.text("Recovered", "text-after-error")]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-missing", "missing", {}), TestLLM.text("Recovered", "text-after-error"))
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(2)
|
|
|
|
@@ -3412,7 +3326,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Call defect")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-defect", "defect", {}), reply.text("Recovered", "text-after-defect")]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-defect", "defect", {}), TestLLM.text("Recovered", "text-after-defect"))
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -3450,9 +3364,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
blocked: ({
|
|
|
|
|
blocked: {
|
|
|
|
|
name: "blocked",
|
|
|
|
|
description: "Fail because policy blocked execution",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
@@ -3461,13 +3376,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.fail(new Permission.BlockedError({ rules: [], permission: "blocked", resources: ["*"] })).pipe(
|
|
|
|
|
Effect.mapError(() => new Tool.Error({ message: "Permission blocked" })),
|
|
|
|
|
),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Call blocked")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-blocked", "blocked", {}), reply.stop()]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-blocked", "blocked", {}), TestLLM.stop())
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -3489,21 +3404,22 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
declined: ({
|
|
|
|
|
declined: {
|
|
|
|
|
name: "declined",
|
|
|
|
|
description: "Fail because the user declined approval",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
|
output: Schema.Struct({}),
|
|
|
|
|
execute: () => Effect.die(new Permission.DeclinedError()),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Call declined")
|
|
|
|
|
|
|
|
|
|
response = reply.tool("call-declined", "declined", {})
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-declined", "declined", {}))
|
|
|
|
|
|
|
|
|
|
const exit = yield* session.resume(sessionID).pipe(Effect.exit)
|
|
|
|
|
|
|
|
|
@@ -3530,9 +3446,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
corrected: ({
|
|
|
|
|
corrected: {
|
|
|
|
|
name: "corrected",
|
|
|
|
|
description: "Fail with user correction feedback",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
@@ -3541,13 +3458,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.fail(new Permission.CorrectedError({ feedback: "Use another tool" })).pipe(
|
|
|
|
|
Effect.mapError(() => new Tool.Error({ message: "Use another tool" })),
|
|
|
|
|
),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Call corrected")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-corrected", "corrected", {}), reply.stop()]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-corrected", "corrected", {}), TestLLM.stop())
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -3571,10 +3488,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
yield* transformTools(registry, { permissionfail: permissionFail }, { codemode: false })
|
|
|
|
|
yield* admit(session, "Reject permission")
|
|
|
|
|
responses = [
|
|
|
|
|
reply.tool("call-permission", "permissionfail", {}),
|
|
|
|
|
[LLMEvent.stepStart({ index: 0 }), LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } })],
|
|
|
|
|
]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-permission", "permissionfail", {}), [
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -3607,21 +3524,22 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
const registry = yield* Tool.Service
|
|
|
|
|
yield* transformTools(registry,
|
|
|
|
|
yield* transformTools(
|
|
|
|
|
registry,
|
|
|
|
|
{
|
|
|
|
|
question: ({
|
|
|
|
|
question: {
|
|
|
|
|
name: "question",
|
|
|
|
|
description: "Ask the user",
|
|
|
|
|
input: Schema.Struct({}),
|
|
|
|
|
output: Schema.Struct({}),
|
|
|
|
|
execute: () => Effect.die(new QuestionTool.CancelledError()),
|
|
|
|
|
}),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
{ codemode: false },
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Ask then stop")
|
|
|
|
|
|
|
|
|
|
responses = [reply.tool("call-question", "question", {}), []]
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-question", "question", {}), [])
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.exit, Effect.forkChild)
|
|
|
|
|
const exit = yield* Fiber.join(run)
|
|
|
|
@@ -3651,12 +3569,14 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* admit(session, "Settle before failing")
|
|
|
|
|
const failure = providerUnavailable()
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-before-failure", name: "echo", input: { text: "settle" } }),
|
|
|
|
|
]),
|
|
|
|
|
Stream.fail(failure),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-before-failure", name: "echo", input: { text: "settle" } }),
|
|
|
|
|
]),
|
|
|
|
|
Stream.fail(failure),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
@@ -3695,12 +3615,14 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Interrupt blocked tool")
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-before-interrupt", name: "echo", input: { text: "blocked" } }),
|
|
|
|
|
]),
|
|
|
|
|
Stream.never,
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-before-interrupt", name: "echo", input: { text: "blocked" } }),
|
|
|
|
|
]),
|
|
|
|
|
Stream.never,
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
@@ -3739,8 +3661,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
{ type: "assistant", content: [{ type: "tool", id: "call-before-interrupt", state: { status: "error" } }] },
|
|
|
|
|
])
|
|
|
|
|
requests.length = 0
|
|
|
|
|
responseStream = undefined
|
|
|
|
|
response = []
|
|
|
|
|
yield* TestLLM.push([])
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
expect(requests[0]?.messages.map((message) => message.role)).toEqual(["user", "assistant", "tool"])
|
|
|
|
|
}),
|
|
|
|
@@ -3750,15 +3671,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Interrupt provider")
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.interrupt(sessionID)
|
|
|
|
|
const exit = yield* Fiber.await(run)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(Exit.isFailure(exit) && Cause.hasInterruptsOnly(exit.cause)).toBeTrue()
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -3778,7 +3696,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsStarted = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsReady = 1
|
|
|
|
|
response = reply.tool("call-await-interrupt", "echo", { text: "blocked" })
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-await-interrupt", "echo", { text: "blocked" }))
|
|
|
|
|
|
|
|
|
|
const runner = yield* SessionRunner.Service
|
|
|
|
|
const run = yield* runner.drain({ sessionID, force: true }).pipe(Effect.forkChild)
|
|
|
|
@@ -3819,10 +3737,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Finish at the limit")
|
|
|
|
|
|
|
|
|
|
responses = [
|
|
|
|
|
reply.tool("call-terminal", "echo", { text: "done" }),
|
|
|
|
|
reply.tool("call-forbidden", "echo", { text: "forbidden" }),
|
|
|
|
|
]
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.tool("call-terminal", "echo", { text: "done" }),
|
|
|
|
|
TestLLM.tool("call-forbidden", "echo", { text: "forbidden" }),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -3855,21 +3773,18 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Start work")
|
|
|
|
|
|
|
|
|
|
responses = [
|
|
|
|
|
reply.tool("call-before-steer", "echo", { text: "before" }),
|
|
|
|
|
reply.tool("call-after-steer", "echo", { text: "after" }),
|
|
|
|
|
reply.stop(),
|
|
|
|
|
]
|
|
|
|
|
streamGate = yield* Deferred.make<void>()
|
|
|
|
|
streamStarted = yield* Deferred.make<void>()
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
TestLLM.tool("call-before-steer", "echo", { text: "before" }),
|
|
|
|
|
TestLLM.tool("call-after-steer", "echo", { text: "after" }),
|
|
|
|
|
TestLLM.stop(),
|
|
|
|
|
)
|
|
|
|
|
const stream = yield* TestLLM.gate
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(streamStarted)
|
|
|
|
|
yield* stream.started
|
|
|
|
|
yield* session.prompt({ sessionID, text: "Change direction" })
|
|
|
|
|
yield* Deferred.succeed(streamGate, undefined)
|
|
|
|
|
yield* stream.release
|
|
|
|
|
yield* Fiber.join(run)
|
|
|
|
|
streamGate = undefined
|
|
|
|
|
streamStarted = undefined
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(3)
|
|
|
|
|
expect(requests[1]?.toolChoice).toBeUndefined()
|
|
|
|
@@ -3884,7 +3799,10 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail durably")
|
|
|
|
|
|
|
|
|
|
response = [LLMEvent.stepStart({ index: 0 }), LLMEvent.providerError({ message: "Provider unavailable" })]
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.providerError({ message: "Provider unavailable" }),
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("Provider unavailable")
|
|
|
|
|
|
|
|
|
@@ -3901,7 +3819,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail before step")
|
|
|
|
|
|
|
|
|
|
response = [LLMEvent.providerError({ message: "Provider unavailable" })]
|
|
|
|
|
yield* TestLLM.push([LLMEvent.providerError({ message: "Provider unavailable" })])
|
|
|
|
|
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("Provider unavailable")
|
|
|
|
|
|
|
|
|
@@ -3917,7 +3835,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Blocked response")
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "partial" }),
|
|
|
|
|
LLMEvent.textDelta({ id: "partial", text: "Partial" }),
|
|
|
|
@@ -3927,7 +3845,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
usage: { nonCachedInputTokens: 8, outputTokens: 3, reasoningTokens: 1 },
|
|
|
|
|
}),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "content-filter" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("Provider blocked the response")
|
|
|
|
|
expect(yield* session.context(sessionID)).toMatchObject([
|
|
|
|
@@ -3956,12 +3874,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsStarted = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsReady = 1
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-before-content-filter", name: "echo", input: { text: "settled" } }),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "content-filter" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "content-filter" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(toolExecutionsStarted)
|
|
|
|
@@ -3989,13 +3907,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail after output")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "text-partial" }),
|
|
|
|
|
LLMEvent.textDelta({ id: "text-partial", text: "Partial" }),
|
|
|
|
|
LLMEvent.textEnd({ id: "text-partial" }),
|
|
|
|
|
LLMEvent.providerError({ message: "prompt too long", classification: "context-overflow" }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("prompt too long")
|
|
|
|
|
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -4016,7 +3934,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail raw stream durably")
|
|
|
|
|
const failure = invalidRequest()
|
|
|
|
|
responseStream = Stream.fail(failure)
|
|
|
|
|
yield* TestLLM.push(Stream.fail(failure))
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
yield* replaySessionProjection(sessionID)
|
|
|
|
@@ -4031,11 +3949,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Retry transport")
|
|
|
|
|
responseStream = Stream.fail(providerUnavailable())
|
|
|
|
|
response = reply.text("Recovered", "retry-success")
|
|
|
|
|
yield* TestLLM.push(Stream.fail(providerUnavailable()))
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Recovered", "retry-success"))
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 1) yield* Effect.yieldNow
|
|
|
|
|
yield* TestLLM.wait(1)
|
|
|
|
|
yield* TestClock.adjust("1999 millis")
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
yield* TestClock.adjust("1 millis")
|
|
|
|
@@ -4058,11 +3976,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Retry rate limit")
|
|
|
|
|
responseStream = Stream.fail(rateLimited(5_000))
|
|
|
|
|
response = reply.text("Recovered", "retry-after-success")
|
|
|
|
|
yield* TestLLM.push(Stream.fail(rateLimited(5_000)))
|
|
|
|
|
yield* TestLLM.push(TestLLM.text("Recovered", "retry-after-success"))
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 1) yield* Effect.yieldNow
|
|
|
|
|
yield* TestLLM.wait(1)
|
|
|
|
|
yield* TestClock.adjust("4999 millis")
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
|
yield* TestClock.adjust("1 millis")
|
|
|
|
@@ -4076,11 +3994,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Do not replay partial output")
|
|
|
|
|
const failure = rateLimited()
|
|
|
|
|
responseStream = Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "partial-rate-limit" }),
|
|
|
|
|
LLMEvent.textDelta({ id: "partial-rate-limit", text: "Partial" }),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure)))
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "partial-rate-limit" }),
|
|
|
|
|
LLMEvent.textDelta({ id: "partial-rate-limit", text: "Partial" }),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure))),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -4101,15 +4021,16 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Exhaust retries")
|
|
|
|
|
streamFailure = providerUnavailable()
|
|
|
|
|
const failure = providerUnavailable()
|
|
|
|
|
yield* TestLLM.always(Stream.fail(failure))
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 1) yield* Effect.yieldNow
|
|
|
|
|
yield* TestLLM.wait(1)
|
|
|
|
|
for (const [index, delay] of [2_000, 4_000, 8_000, 16_000].entries()) {
|
|
|
|
|
yield* TestClock.adjust(delay)
|
|
|
|
|
while (requests.length < index + 2) yield* Effect.yieldNow
|
|
|
|
|
yield* TestLLM.wait(index + 2)
|
|
|
|
|
}
|
|
|
|
|
expect(yield* Fiber.join(run).pipe(Effect.flip)).toBe(streamFailure)
|
|
|
|
|
expect(yield* Fiber.join(run).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
expect(requests).toHaveLength(5)
|
|
|
|
|
|
|
|
|
|
const database = (yield* Database.Service).db
|
|
|
|
@@ -4150,11 +4071,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
)
|
|
|
|
|
yield* admit(session, "Retry without consuming a step")
|
|
|
|
|
const failure = providerUnavailable()
|
|
|
|
|
responseStream = Stream.fail(failure)
|
|
|
|
|
responses = [reply.tool("call-after-retry", "echo", { text: "recovered" }), reply.stop()]
|
|
|
|
|
yield* TestLLM.push(Stream.fail(failure))
|
|
|
|
|
yield* TestLLM.push(TestLLM.tool("call-after-retry", "echo", { text: "recovered" }), TestLLM.stop())
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
while (requests.length < 1) yield* Effect.yieldNow
|
|
|
|
|
yield* TestLLM.wait(1)
|
|
|
|
|
yield* TestClock.adjust("2 seconds")
|
|
|
|
|
yield* Fiber.join(run)
|
|
|
|
|
|
|
|
|
@@ -4187,7 +4108,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Do not retry")
|
|
|
|
|
const failure = invalidRequest()
|
|
|
|
|
streamFailure = failure
|
|
|
|
|
yield* TestLLM.push(Stream.fail(failure))
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
expect(requests).toHaveLength(1)
|
|
|
|
@@ -4204,16 +4125,18 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
method: "stream",
|
|
|
|
|
reason: new InvalidProviderOutputReason({ message: "Invalid JSON input for tool call echo" }),
|
|
|
|
|
})
|
|
|
|
|
responseStream = Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputStart({ id: "call-malformed", name: "echo" }),
|
|
|
|
|
LLMEvent.toolInputDelta({ id: "call-malformed", name: "echo", text: '{"text":"partial' }),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure)))
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputStart({ id: "call-malformed", name: "echo" }),
|
|
|
|
|
LLMEvent.toolInputDelta({ id: "call-malformed", name: "echo", text: '{"text":"partial' }),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure))),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
const assistant = requireAssistant(yield* session.context(sessionID))
|
|
|
|
|
|
|
|
|
|
response = reply.stop()
|
|
|
|
|
yield* TestLLM.push(TestLLM.stop())
|
|
|
|
|
yield* admit(session, "Continue")
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4240,7 +4163,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
yield* admit(session, "Recover malformed tool input")
|
|
|
|
|
const marker = "raw-malformed-marker"
|
|
|
|
|
const raw = `{"text":"${marker}`
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputStart({ id: "call-malformed", name: "echo" }),
|
|
|
|
@@ -4254,8 +4177,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
],
|
|
|
|
|
reply.stop(),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.stop(),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4339,7 +4262,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsStarted = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsReady = 1
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
[
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-valid", name: "echo", input: { text: "valid" } }),
|
|
|
|
@@ -4351,8 +4274,8 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
],
|
|
|
|
|
reply.stop(),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.stop(),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(toolExecutionsStarted)
|
|
|
|
@@ -4382,7 +4305,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsStarted = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsReady = 1
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-valid", name: "echo", input: { text: "blocked" } }),
|
|
|
|
|
LLMEvent.toolInputError({
|
|
|
|
@@ -4392,7 +4315,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
}),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(toolExecutionsStarted)
|
|
|
|
@@ -4433,11 +4356,13 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
method: "stream",
|
|
|
|
|
reason: new InvalidProviderOutputReason({ message: "Invalid hosted tool input" }),
|
|
|
|
|
})
|
|
|
|
|
responseStream = Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputStart({ id: "call-hosted", name: "web_search", providerExecuted: true }),
|
|
|
|
|
LLMEvent.toolInputDelta({ id: "call-hosted", name: "web_search", text: '{"query":"partial' }),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure)))
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputStart({ id: "call-hosted", name: "web_search", providerExecuted: true }),
|
|
|
|
|
LLMEvent.toolInputDelta({ id: "call-hosted", name: "web_search", text: '{"query":"partial' }),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure))),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
expect(requireAssistant(yield* session.context(sessionID))).toMatchObject({
|
|
|
|
@@ -4463,14 +4388,16 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
method: "stream",
|
|
|
|
|
reason: new InvalidProviderOutputReason({ message: "Provider failed after malformed input" }),
|
|
|
|
|
})
|
|
|
|
|
responseStream = Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputError({
|
|
|
|
|
id: "call-malformed",
|
|
|
|
|
name: "echo",
|
|
|
|
|
raw: '{"text":"partial',
|
|
|
|
|
}),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure)))
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputError({
|
|
|
|
|
id: "call-malformed",
|
|
|
|
|
name: "echo",
|
|
|
|
|
raw: '{"text":"partial',
|
|
|
|
|
}),
|
|
|
|
|
]).pipe(Stream.concat(Stream.fail(failure))),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
|
expect(requireAssistant(yield* session.context(sessionID))).toMatchObject({
|
|
|
|
@@ -4502,12 +4429,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
]
|
|
|
|
|
responses = [
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
malformed("call-first"),
|
|
|
|
|
reply.tool("call-valid-between", "echo", { text: "valid" }),
|
|
|
|
|
TestLLM.tool("call-valid-between", "echo", { text: "valid" }),
|
|
|
|
|
malformed("call-second"),
|
|
|
|
|
reply.stop(),
|
|
|
|
|
]
|
|
|
|
|
TestLLM.stop(),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4537,7 +4464,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "tool-calls" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "tool-calls" } }),
|
|
|
|
|
]
|
|
|
|
|
responses = [malformed("call-first"), malformed("call-at-limit")]
|
|
|
|
|
yield* TestLLM.push(malformed("call-first"), malformed("call-at-limit"))
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4556,11 +4483,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsStarted = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionsReady = 1
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-before-provider-error", name: "echo", input: { text: "settled" } }),
|
|
|
|
|
LLMEvent.providerError({ message: "Provider unavailable" }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
|
yield* Deferred.await(toolExecutionsStarted)
|
|
|
|
@@ -4587,11 +4514,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail hosted tool durably")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
hostedCall("call-hosted-provider-error", "effect"),
|
|
|
|
|
LLMEvent.providerError({ message: "Provider unavailable" }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("Provider unavailable")
|
|
|
|
|
|
|
|
|
@@ -4618,11 +4545,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Defect while provider fails")
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-defect-provider-error", name: "defect", input: {} }),
|
|
|
|
|
LLMEvent.providerError({ message: "Provider unavailable" }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("Provider unavailable")
|
|
|
|
|
|
|
|
|
@@ -4643,11 +4570,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Storage fails while provider fails")
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-store-provider-error", name: "storefail", input: {} }),
|
|
|
|
|
LLMEvent.providerError({ message: "Provider unavailable" }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.exit)).toMatchObject({ _tag: "Failure" })
|
|
|
|
|
|
|
|
|
@@ -4661,7 +4588,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail hosted tool at EOF")
|
|
|
|
|
response = [LLMEvent.stepStart({ index: 0 }), hostedCall("call-hosted-eof", "effect")]
|
|
|
|
|
yield* TestLLM.push([LLMEvent.stepStart({ index: 0 }), hostedCall("call-hosted-eof", "effect")])
|
|
|
|
|
|
|
|
|
|
expect((yield* session.resume(sessionID).pipe(Effect.flip)).message).toBe("Provider did not return a tool result")
|
|
|
|
|
const assistant = requireAssistant(yield* session.context(sessionID))
|
|
|
|
@@ -4693,12 +4620,12 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Settle hosted tool before ending")
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
hostedCall("call-hosted-clean-end", "effect"),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4723,13 +4650,17 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const failure = invalidRequest()
|
|
|
|
|
const providerFailed = yield* Deferred.make<void>()
|
|
|
|
|
toolExecutionGate = yield* Deferred.make<void>()
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-local-raw-failure", name: "defect", input: {} }),
|
|
|
|
|
hostedCall("call-hosted-raw-failure-pair", "effect"),
|
|
|
|
|
]),
|
|
|
|
|
Stream.fromEffect(Deferred.succeed(providerFailed, undefined)).pipe(Stream.flatMap(() => Stream.fail(failure))),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(
|
|
|
|
|
Stream.fromIterable([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolCall({ id: "call-local-raw-failure", name: "defect", input: {} }),
|
|
|
|
|
hostedCall("call-hosted-raw-failure-pair", "effect"),
|
|
|
|
|
]),
|
|
|
|
|
Stream.fromEffect(Deferred.succeed(providerFailed, undefined)).pipe(
|
|
|
|
|
Stream.flatMap(() => Stream.fail(failure)),
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
|
|
|
|
@@ -4759,9 +4690,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Fail hosted tool on raw failure")
|
|
|
|
|
const failure = providerUnavailable()
|
|
|
|
|
responseStream = Stream.concat(
|
|
|
|
|
Stream.fromIterable([LLMEvent.stepStart({ index: 0 }), hostedCall("call-hosted-raw-failure", "effect")]),
|
|
|
|
|
Stream.fail(failure),
|
|
|
|
|
yield* TestLLM.push(
|
|
|
|
|
Stream.concat(
|
|
|
|
|
Stream.fromIterable([LLMEvent.stepStart({ index: 0 }), hostedCall("call-hosted-raw-failure", "effect")]),
|
|
|
|
|
Stream.fail(failure),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
|
|
|
|
@@ -4795,11 +4728,11 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Two blocks")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "text-1" }),
|
|
|
|
|
LLMEvent.textStart({ id: "text-2" }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
const defect = yield* session.resume(sessionID).pipe(Effect.catchDefect(Effect.succeed))
|
|
|
|
|
expect(defect).toBeInstanceOf(Error)
|
|
|
|
@@ -4813,7 +4746,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Two blocks")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.textStart({ id: "text-1" }),
|
|
|
|
|
LLMEvent.textDelta({ id: "text-1", text: "First" }),
|
|
|
|
@@ -4823,7 +4756,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
LLMEvent.textEnd({ id: "text-2" }),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4855,7 +4788,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("rejects duplicate streamed text starts", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = [LLMEvent.textStart({ id: "text-1" }), LLMEvent.textStart({ id: "text-1" })]
|
|
|
|
|
yield* TestLLM.push([LLMEvent.textStart({ id: "text-1" }), LLMEvent.textStart({ id: "text-1" })])
|
|
|
|
|
|
|
|
|
|
const defect = yield* session.resume(sessionID).pipe(Effect.catchDefect(Effect.succeed))
|
|
|
|
|
expect(defect).toBeInstanceOf(Error)
|
|
|
|
@@ -4869,7 +4802,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
yield* admit(session, "Call provider tool")
|
|
|
|
|
|
|
|
|
|
response = [
|
|
|
|
|
yield* TestLLM.push([
|
|
|
|
|
LLMEvent.stepStart({ index: 0 }),
|
|
|
|
|
LLMEvent.toolInputStart({ id: "call-parsed", name: "web_search" }),
|
|
|
|
|
LLMEvent.toolInputDelta({ id: "call-parsed", name: "web_search", text: '{"query":"hello"}' }),
|
|
|
|
@@ -4877,7 +4810,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
hostedCall("call-parsed", "hello"),
|
|
|
|
|
LLMEvent.stepFinish({ index: 0, reason: { normalized: "stop" } }),
|
|
|
|
|
LLMEvent.finish({ reason: { normalized: "stop" } }),
|
|
|
|
|
]
|
|
|
|
|
])
|
|
|
|
|
|
|
|
|
|
yield* session.resume(sessionID)
|
|
|
|
|
|
|
|
|
@@ -4894,7 +4827,7 @@ describe("SessionRunnerLLM", () => {
|
|
|
|
|
it.effect("rejects malformed streamed tool input ordering", () =>
|
|
|
|
|
Effect.gen(function* () {
|
|
|
|
|
const session = yield* setup
|
|
|
|
|
response = [LLMEvent.toolInputDelta({ id: "call-1", name: "read", text: "{}" })]
|
|
|
|
|
yield* TestLLM.push([LLMEvent.toolInputDelta({ id: "call-1", name: "read", text: "{}" })])
|
|
|
|
|
|
|
|
|
|
const defect = yield* session.resume(sessionID).pipe(Effect.catchDefect(Effect.succeed))
|
|
|
|
|
expect(defect).toBeInstanceOf(Error)
|
|
|
|
|