diff --git a/packages/llm/DESIGN.http-retry.md b/packages/llm/DESIGN.http-retry.md new file mode 100644 index 0000000000..182e081a11 --- /dev/null +++ b/packages/llm/DESIGN.http-retry.md @@ -0,0 +1,342 @@ +# LLM HTTP Diagnostics And Retry Plan + +## Goal + +Improve provider HTTP failures so they are easier to debug, safer to report, and retryable only at boundaries that do not replay a partially consumed model stream. + +The first implementation should prioritize diagnostics and conservative rate-limit / overload retries. Transport retries for generation `POST`s are ambiguous because a timeout or connection reset does not prove the provider did not receive and process the request. + +## Current State + +`src/adapter/executor.ts` centralizes provider HTTP execution through `RequestExecutor.Service`: + +```ts +execute: (request) => http.execute(request).pipe(Effect.mapError(toHttpError), Effect.flatMap(statusError)) +``` + +Current typed failures are intentionally small: + +- `ProviderRequestError`: HTTP status, message, optional body. +- `TransportError`: message, optional reason, optional URL. + +This is enough for coarse handling, but weak for production debugging and retry decisions. A failed request does not carry redacted request headers, response headers, provider request IDs, retry hints, or parsed `Retry-After` timing. + +## Non-Goals + +- Do not retry after any response stream element has been returned to an adapter parser. +- Do not retry provider chunk parse errors or mid-stream provider error events. +- Do not add provider-specific error classes in the first pass. +- Do not parse every provider error body into provider-native shapes in the executor. +- Do not add broad replay semantics for tool loops, provider-executed tools, or partial generations. +- Do not expose secrets in error values, logs, snapshots, or tests. + +## Design + +### 1. Add HTTP Diagnostic Shapes + +Add reusable schema classes in `src/schema.ts`: + +```ts +export class HttpRequestDetails extends Schema.Class("LLM.HttpRequestDetails")({ + method: Schema.String, + url: Schema.String, + headers: Schema.Record(Schema.String, Schema.String), +}) {} + +export class HttpResponseDetails extends Schema.Class("LLM.HttpResponseDetails")({ + status: Schema.Number, + headers: Schema.Record(Schema.String, Schema.String), +}) {} +``` + +Extend `ProviderRequestError`: + +```ts +export class ProviderRequestError extends Schema.TaggedErrorClass()("LLM.ProviderRequestError", { + status: Schema.Number, + message: Schema.String, + body: Schema.optional(Schema.String), + bodyTruncated: Schema.optional(Schema.Boolean), + retryable: Schema.optional(Schema.Boolean), + retryAfterMs: Schema.optional(Schema.Number), + requestId: Schema.optional(Schema.String), + rateLimit: Schema.optional(HttpRateLimitDetails), + request: Schema.optional(HttpRequestDetails), + response: Schema.optional(HttpResponseDetails), +}) {} +``` + +Extend `TransportError` for diagnostics, but do not make transport retry automatic in the first patch: + +```ts +export class TransportError extends Schema.TaggedErrorClass()("LLM.TransportError", { + message: Schema.String, + reason: Schema.optional(Schema.String), + url: Schema.optional(Schema.String), + retryable: Schema.optional(Schema.Boolean), + request: Schema.optional(HttpRequestDetails), +}) {} +``` + +Add a small normalized rate-limit shape if it remains simple: + +```ts +export class HttpRateLimitDetails extends Schema.Class("LLM.HttpRateLimitDetails")({ + retryAfterMs: Schema.optional(Schema.Number), + limit: Schema.optional(Schema.String), + remaining: Schema.optional(Schema.String), + reset: Schema.optional(Schema.String), +}) {} +``` + +If `HttpRateLimitDetails` starts becoming provider-specific, skip it in the first patch and rely on redacted response headers plus `retryAfterMs`. + +### 2. Redact Headers, URLs, And Bodies + +Redaction must happen before typed errors are constructed. + +Prefer Effect's redaction context if it is convenient from `effect/unstable/http`: + +- Extend `Headers.CurrentRedactedNames` with package-sensitive names. +- Use the equivalent of `Redactable.redact(...)` for request and response headers. + +Keep a local matcher for URL query parameters and as a fallback policy: + +```ts +const sensitiveName = (name: string) => + /authorization|api[-_]?key|token|secret|credential|signature|x-amz-signature/i.test(name) +``` + +Header redaction: + +```ts +const redactHeaders = (headers: Record) => + Object.fromEntries( + Object.entries(headers).map(([name, value]) => [name, sensitiveName(name) ? "" : value]), + ) +``` + +URL redaction: + +```ts +const redactUrl = (value: string) => { + const url = new URL(value) + url.searchParams.forEach((_, key) => { + if (sensitiveName(key)) url.searchParams.set(key, "") + }) + return url.toString() +} +``` + +Response body handling: + +- Cap stored bodies, for example at `16_384` characters. +- Set `bodyTruncated: true` when capped. +- Do not attempt deep provider-specific body redaction in the first pass unless a known secret field is easy to scrub safely. +- Consider reusing the HTTP recorder's secret scanning helpers if they are package-accessible without making `llm` tests depend on recorder internals. + +### 3. Extract Request, Response, And Provider Request IDs + +`statusError` must receive the original request. The current shape `statusError(response)` cannot populate request diagnostics reliably. + +Use a closure: + +```ts +const statusError = (request: HttpClientRequest.HttpClientRequest) => + (response: HttpClientResponse.HttpClientResponse) => + Effect.gen(function* () { + if (response.status < 400) return response + // construct ProviderRequestError with request + response diagnostics + }) +``` + +Or switch to `HttpClient.filterStatusOk` and map the resulting `StatusCodeError`, which carries both request and response. The closure approach is the smaller change against the current executor. + +Normalize headers once for case-insensitive lookups: + +```ts +const normalizedHeaders = (headers: Record) => + Object.fromEntries(Object.entries(headers).map(([key, value]) => [key.toLowerCase(), value])) +``` + +Request ID extraction should be conservative and provider-agnostic: + +```ts +const requestId = (headers: Record) => { + const normalized = normalizedHeaders(headers) + return normalized["x-request-id"] ?? + normalized["request-id"] ?? + normalized["x-amzn-requestid"] ?? + normalized["x-amz-request-id"] ?? + normalized["x-goog-request-id"] ?? + normalized["cf-ray"] +} +``` + +This is diagnostic only; adapters can still expose richer provider metadata later. + +### 4. Classify Retryable Status Responses Conservatively + +Automatic retry should initially apply only to explicit HTTP status responses where no model stream was handed to a parser. + +Default automatic retry statuses: + +- `429 Too Many Requests` +- `503 Service Unavailable` +- `504 Gateway Timeout` +- `529 Overloaded` used by Anthropic-style overload responses + +Do not include `409` in provider-neutral defaults. Effect-smol treats OpenAI `409` as invalid request-like behavior, and there is not enough provider evidence to retry it globally. + +Do not automatically retry transport timeouts / connection resets in the first patch. Marking them as diagnostically retryable can be considered later behind explicit opt-in, but default generation retries should not replay ambiguous `POST`s. + +Implementation helper: + +```ts +const retryableStatus = (status: number) => + status === 429 || status === 503 || status === 504 || status === 529 +``` + +Potential future additions after provider evidence: + +- `500`, `502` for transient provider failures. +- Cloudflare edge statuses such as `520`, `522`, `524` for OpenAI-compatible front doors. +- Provider-specific policies keyed by adapter/provider. + +### 5. Parse `Retry-After` And Simple Rate-Limit Headers + +Parse standard `Retry-After` forms: + +- Delta seconds: `Retry-After: 3` +- HTTP date: `Retry-After: Wed, 21 Oct 2015 07:28:00 GMT` + +Also accept `retry-after-ms` when present. + +```ts +const retryAfterMs = (headers: Record) => { + const normalized = normalizedHeaders(headers) + const millis = Number(normalized["retry-after-ms"]) + if (Number.isFinite(millis)) return Math.max(0, millis) + + const value = normalized["retry-after"] + if (!value) return undefined + + const seconds = Number(value) + if (Number.isFinite(seconds)) return Math.max(0, seconds * 1000) + + const date = Date.parse(value) + if (!Number.isNaN(date)) return Math.max(0, date - Date.now()) + + return undefined +} +``` + +Keep raw redacted headers on `HttpResponseDetails` so callers can inspect provider-specific rate-limit headers such as `x-ratelimit-*`, `anthropic-ratelimit-*`, or AWS/Gemini equivalents without the executor knowing every provider shape. + +### 6. Add Conservative Pre-Stream Retry In `RequestExecutor` + +Retry should live in `src/adapter/executor.ts`, not in each adapter. + +The executor owns this boundary: + +```txt +compile request -> execute HTTP request -> receive response -> parse stream +``` + +Automatic retry is allowed only before `execute` returns a successful response. After that, stream consumers own the response and retrying could duplicate text, tool calls, hosted tool side effects, or token charges. + +Default retry policy: + +- `maxRetries`: `2` +- Base delay: `500ms` +- Max delay: `10s` +- Jitter: enabled when no `retryAfterMs` is present +- Honor `retryAfterMs` when present, capped by max delay in the first patch +- Retry predicate: only `ProviderRequestError` with `retryable === true` + +Use Effect scheduling primitives if the v4 API can express error-dependent delay cleanly. If not, keep a small private helper rather than exposing retry machinery publicly. + +The shape should be similar to: + +```ts +const executeOnce = (request: HttpClientRequest.HttpClientRequest) => + http.execute(request).pipe( + Effect.mapError(toHttpError), + Effect.flatMap(statusError(request)), + ) + +execute: (request) => executeOnce(request).pipe(retryStatusFailures(defaultRetryPolicy)) +``` + +`retryStatusFailures` should stay private until there is a concrete external need. + +### 7. Future Retry Configuration Requires Executor Context + +Do not add `HttpOptions.retry` in the first patch. + +`RequestExecutor.execute` currently receives only `HttpClientRequest.HttpClientRequest`. It does not receive the original `LLMRequest`, merged model/request `HttpOptions`, adapter ID, provider ID, or generation/tool context. + +Per-request retry configuration requires one of these changes first: + +```ts +execute: (input: { + readonly http: HttpClientRequest.HttpClientRequest + readonly request: LLMRequest +}) => Effect.Effect +``` + +or: + +```ts +execute: ( + http: HttpClientRequest.HttpClientRequest, + context: RequestExecutor.Context, +) => Effect.Effect +``` + +Defer that API change until default diagnostics and conservative status retry are proven useful. + +## Implementation Plan + +1. Add `HttpRequestDetails` and `HttpResponseDetails` schema classes. +2. Optionally add `HttpRateLimitDetails` if it stays provider-neutral. +3. Extend `ProviderRequestError` and `TransportError` with diagnostics and retry hints. +4. Add executor helpers for header normalization, redaction, URL redaction, body truncation, request details, response details, request IDs, retryable status classification, and `Retry-After` parsing. +5. Change `statusError(response)` to `statusError(request)(response)` or equivalent so rich request diagnostics are available. +6. Populate rich `ProviderRequestError` for non-2xx status responses. +7. Populate richer `TransportError` where the underlying HTTP client error exposes a request, but do not retry transport errors by default. +8. Add private conservative retry around `executeOnce` for retryable status responses only. +9. Add deterministic tests for diagnostics, redaction, `Retry-After`, retryable statuses, non-retryable statuses, retry attempts, and no retry after stream parsing begins. + +## Tests + +Add or extend tests under `packages/llm/test`: + +- A `429` response returns `ProviderRequestError` with `retryable: true`, parsed `retryAfterMs`, redacted request headers, redacted response headers, redacted URL query secrets, and request ID. +- A `529` response is treated as retryable. +- A `401` response returns `ProviderRequestError` with `retryable: false` or `undefined`, not retried. +- A `503` followed by a successful SSE response retries exactly once and streams normally. +- A repeated `429` retries up to the default limit, then returns the final enriched error. +- Authorization-like request headers are redacted in the error. +- Query-string secrets are redacted in `request.url`. +- Non-secret headers remain visible for diagnostics. +- Response bodies are truncated and set `bodyTruncated: true` when above the cap. +- Transport timeout or connection errors become `TransportError` diagnostics but are not retried by default. +- Invalid URL or encode failures become `TransportError` with `retryable: false` or `undefined`. +- A first response of `200` with one valid SSE event followed by malformed data is attempted exactly once and fails as a stream/chunk parse error, proving executor retry does not replay partial streams. + +Use deterministic scripted HTTP responses over live provider calls. Use a controlled clock or a test-only short retry policy so retry tests are not slow or flaky. Do not add recorded cassettes for retry behavior unless a real provider behavior must be captured. + +## Open Questions + +- Should explicit `Retry-After` be allowed to exceed `maxDelayMs`, or should the first implementation cap it for responsiveness? +- Should response body redaction go beyond truncation in the first patch, and can recorder secret scanning be reused safely? +- Should `ProviderRequestError` distinguish `rateLimited: true` from generic `retryable: true`, or is `status === 429` sufficient? +- Should default retry later include `500`, `502`, `520`, `522`, or `524` after OpenAI-compatible provider evidence? +- Should ambiguous transport retries be opt-in through a future executor context once the API can see provider/model/request settings? + +## Recommended First Patch Boundary + +Include diagnostics, redaction for headers and URL query params, response body truncation, request ID extraction, conservative retry classification, `Retry-After` parsing, and default pre-stream retries for explicit rate-limit / overload status responses. + +Defer provider-specific error body parsing, public retry configuration, ambiguous transport retries, and broad 5xx retry defaults until after the executor behavior is tested against OpenAI, Anthropic, Gemini, OpenAI-compatible providers, and Bedrock deterministic fixtures. diff --git a/packages/llm/example/tutorial.ts b/packages/llm/example/tutorial.ts index 1b654d4811..e93545e55e 100644 --- a/packages/llm/example/tutorial.ts +++ b/packages/llm/example/tutorial.ts @@ -1,5 +1,5 @@ import { Effect, Formatter, Layer, Schema, Stream } from "effect" -import { LLM, LLMClient, Tool } from "@opencode-ai/llm" +import { LLM, LLMClient, Tool, ToolRuntime } from "@opencode-ai/llm" import { Adapter, Auth, Endpoint, Framing, Protocol, RequestExecutor } from "@opencode-ai/llm/adapter" import { OpenAI } from "@opencode-ai/llm/providers" @@ -27,9 +27,8 @@ const model = OpenAI.model("gpt-4o-mini", { }, }) -// 2. Build a provider-neutral request. This is optional for one-off calls — the -// same fields can be passed directly to `LLM.generate` / `LLM.stream` — but it -// is useful when reusing one request across generate and stream examples. +// 2. Build a provider-neutral request. This is useful when reusing one request +// across generate and stream examples. // // Options can live on both the model and the request: // @@ -67,7 +66,8 @@ const rawOverlayExample = LLM.request({ // 3. `generate` sends the request and collects the event stream into one // response object. `response.text` is the collected text output. const generateOnce = Effect.gen(function* () { - const response = yield* LLM.generate(request) + const client = yield* LLMClient.Service + const response = yield* client.generate(request) console.log("\n== generate ==") console.log("generated text:", response.text) @@ -76,20 +76,23 @@ const generateOnce = Effect.gen(function* () { // 4. `stream` exposes provider output as common `LLMEvent`s for UIs that want // incremental text, reasoning, tool input, usage, or finish events. -const streamText = LLM.stream(request).pipe( - Stream.tap((event) => - Effect.sync(() => { - if (event.type === "text-delta") process.stdout.write(`\ntext: ${event.text}`) - if (event.type === "request-finish") process.stdout.write(`\nfinish: ${event.reason}\n`) - }), - ), - Stream.runDrain, -) +const streamText = Effect.gen(function* () { + const client = yield* LLMClient.Service + return yield* client.stream(request).pipe( + Stream.tap((event) => + Effect.sync(() => { + if (event.type === "text-delta") process.stdout.write(`\ntext: ${event.text}`) + if (event.type === "request-finish") process.stdout.write(`\nfinish: ${event.reason}\n`) + }), + ), + Stream.runDrain, + ) +}) -// 5. Tools are typed with Effect Schema. `streamWithTools` adds tool definitions -// to the request, dispatches matching tool calls, validates handler output, -// appends tool results to the next model round, and stops on a final non-tool -// response. +// 5. Tools are typed with Effect Schema. `ToolRuntime.Service` adds tool +// definitions to the request, dispatches matching tool calls, validates handler +// output, appends tool results to the next model round, and stops on a final +// non-tool response. const tools = { get_weather: Tool.make({ description: "Get current weather for a city.", @@ -99,22 +102,27 @@ const tools = { }), } -const streamWithTools = LLM.streamWithTools({ - model, - prompt: "Use get_weather for San Francisco, then answer in one sentence.", - generation: { maxTokens: 80, temperature: 0 }, - tools, - maxSteps: 3, -}).pipe( - Stream.tap((event) => - Effect.sync(() => { - if (event.type === "tool-call") console.log("tool call", event.name, event.input) - if (event.type === "tool-result") console.log("tool result", event.name, event.result) - if (event.type === "text-delta") process.stdout.write(event.text) +const streamWithTools = Effect.gen(function* () { + const runtime = yield* ToolRuntime.Service + return yield* runtime.run({ + request: LLM.request({ + model, + prompt: "Use get_weather for San Francisco, then answer in one sentence.", + generation: { maxTokens: 80, temperature: 0 }, }), - ), - Stream.runDrain, -) + tools, + maxSteps: 3, + }).pipe( + Stream.tap((event) => + Effect.sync(() => { + if (event.type === "tool-call") console.log("tool call", event.name, event.input) + if (event.type === "tool-result") console.log("tool result", event.name, event.result) + if (event.type === "text-delta") process.stdout.write(event.text) + }), + ), + Stream.runDrain, + ) +}) // ----------------------------------------------------------------------------- // Part 2: provider composition with a fake provider @@ -172,7 +180,8 @@ const FakeEcho = { // payload conversion, validation, endpoint, auth, and HTTP construction without // sending anything over the network. const inspectFakeProvider = Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const client = yield* LLMClient.Service + const prepared = yield* client.prepare( LLM.request({ model: FakeEcho.model("tiny-echo"), prompt: "Show me the provider pipeline.", @@ -187,12 +196,23 @@ const inspectFakeProvider = Effect.gen(function* () { // Provide the LLM runtime and the HTTP request executor once. Keep one path // enabled at a time so the tutorial can demonstrate generate, prepare, stream, // or tool-loop behavior without spending tokens on every example. +const requestExecutorLayer = RequestExecutor.defaultLayer +const llmClientLayer = LLMClient.layer.pipe(Layer.provide(requestExecutorLayer)) + const program = Effect.gen(function* () { // yield* generateOnce // yield* inspectFakeProvider - // yield* LLMClient.prepare(rawOverlayExample).pipe(Effect.andThen((prepared) => Effect.sync(() => console.log(prepared.payload)))) + // yield* (yield* LLMClient.Service).prepare(rawOverlayExample).pipe(Effect.andThen((prepared) => Effect.sync(() => console.log(prepared.payload)))) // yield* streamText yield* streamWithTools -}).pipe(Effect.provide(LLM.layer.pipe(Layer.provide(RequestExecutor.defaultLayer)))) +}).pipe( + Effect.provide( + Layer.mergeAll( + requestExecutorLayer, + llmClientLayer, + ToolRuntime.layer.pipe(Layer.provide(llmClientLayer)), + ), + ), +) Effect.runPromise(program) diff --git a/packages/llm/src/adapter/auth.ts b/packages/llm/src/adapter/auth.ts index cbe2a372f6..3c52986f2a 100644 --- a/packages/llm/src/adapter/auth.ts +++ b/packages/llm/src/adapter/auth.ts @@ -1,4 +1,5 @@ import { Effect } from "effect" +import { Headers } from "effect/unstable/http" import type { LLMError, LLMRequest } from "../schema" /** @@ -17,14 +18,14 @@ import type { LLMError, LLMRequest } from "../schema" * future Azure AAD) implement `Auth` as a function that hashes the body, * mints a signature, and merges signed headers into the result. */ -export type Auth = (input: AuthInput) => Effect.Effect, LLMError> +export type Auth = (input: AuthInput) => Effect.Effect export interface AuthInput { readonly request: LLMRequest readonly method: "POST" | "GET" readonly url: string readonly body: string - readonly headers: Record + readonly headers: Headers.Headers } /** @@ -40,11 +41,13 @@ export const passthrough: Auth = ({ headers }) => Effect.succeed(headers) * `model.apiKey` is unset, so callers who pre-set their own auth header keep * working. The shared core for `bearer` and `apiKeyHeader`. */ -const fromApiKey = (from: (apiKey: string) => Record): Auth => ({ request, headers }) => { - const key = request.model.apiKey - if (!key) return Effect.succeed(headers) - return Effect.succeed({ ...headers, ...from(key) }) -} +const fromApiKey = + (from: (apiKey: string) => Headers.Input): Auth => + ({ request, headers }) => { + const key = request.model.apiKey + if (!key) return Effect.succeed(headers) + return Effect.succeed(Headers.setAll(headers, from(key))) + } /** * `Authorization: Bearer ` from `request.model.apiKey`. No-op when @@ -61,12 +64,9 @@ export const openAI: Auth = ({ request, headers }) => { const key = request.model.apiKey if (!key) return Effect.succeed(headers) if (request.model.provider === "azure") { - return Effect.succeed({ - ...Object.fromEntries(Object.entries(headers).filter(([name]) => name.toLowerCase() !== "authorization")), - "api-key": key, - }) + return Effect.succeed(Headers.set(Headers.remove(headers, "authorization"), "api-key", key)) } - return Effect.succeed({ ...headers, authorization: `Bearer ${key}` }) + return Effect.succeed(Headers.set(headers, "authorization", `Bearer ${key}`)) } /** diff --git a/packages/llm/src/adapter/client.ts b/packages/llm/src/adapter/client.ts index 23dbd020a6..0689921fa1 100644 --- a/packages/llm/src/adapter/client.ts +++ b/packages/llm/src/adapter/client.ts @@ -1,5 +1,5 @@ -import { Effect, Schema, Stream } from "effect" -import { HttpClientRequest, type HttpClientResponse } from "effect/unstable/http" +import { Context, Effect, Layer, Schema, Stream } from "effect" +import { Headers, HttpClientRequest, type HttpClientResponse } from "effect/unstable/http" import type { Auth } from "./auth" import { bearer as authBearer } from "./auth" import { type Endpoint, render as renderEndpoint } from "./endpoint" @@ -183,7 +183,7 @@ function model( } } -export interface LLMClient { +export interface Interface { /** * Compile a request through protocol payload lowering, validation, and HTTP * construction without sending it. Returns the prepared request including the @@ -195,10 +195,12 @@ export interface LLMClient { * adapter the request will resolve to. */ readonly prepare: (request: LLMRequest) => Effect.Effect, LLMError> - readonly stream: (request: LLMRequest) => Stream.Stream - readonly generate: (request: LLMRequest) => Effect.Effect + readonly stream: (request: LLMRequest) => Stream.Stream + readonly generate: (request: LLMRequest) => Effect.Effect } +export class Service extends Context.Service()("@opencode/LLMClient") {} + const noAdapter = (model: ModelRef) => new NoAdapterError({ adapter: model.adapter, protocol: model.protocol, provider: model.provider, model: model.id }) @@ -281,7 +283,11 @@ export function make( : ProviderShared.isRecord(payload) ? ProviderShared.encodeJson(mergeJsonRecords(payload, ctx.request.http.body) ?? {}) : yield* ProviderShared.invalidRequest("http.body can only overlay JSON object request bodies") - const merged = { ...buildHeaders({ request: ctx.request }), ...ctx.request.model.headers, ...ctx.request.http?.headers } + const merged = Headers.fromInput({ + ...buildHeaders({ request: ctx.request }), + ...ctx.request.model.headers, + ...ctx.request.http?.headers, + }) const headers = yield* auth({ request: ctx.request, method: "POST", @@ -347,18 +353,17 @@ const prepare = Effect.fn("LLMClient.prepare")(function* (request: LLMRequest) { }) }) -const stream = (request: LLMRequest) => +const streamWith = (executor: RequestExecutor.Interface) => (request: LLMRequest) => Stream.unwrap( Effect.gen(function* () { const compiled = yield* compile(request) - const executor = yield* RequestExecutor.Service const response = yield* executor.execute(compiled.http) return compiled.adapter.parse(response, { request: compiled.request }) }), ) -const generate = Effect.fn("LLM.generate")(function* (request: LLMRequest) { +const generateWith = (stream: Interface["stream"]) => Effect.fn("LLM.generate")(function* (request: LLMRequest) { return new LLMResponse( yield* stream(request).pipe( Stream.runFold( @@ -373,9 +378,17 @@ const generate = Effect.fn("LLM.generate")(function* (request: LLMRequest) { ) }) +export const layer: Layer.Layer = Layer.effect( + Service, + Effect.gen(function* () { + const stream = streamWith(yield* RequestExecutor.Service) + return Service.of({ prepare: prepare as Interface["prepare"], stream, generate: generateWith(stream) }) + }), +) + export const Adapter = { make, model } as const -// The runtime always emits a `PreparedRequest` (payload: unknown). Callers who -// supply a `Payload` type argument assert the shape they expect from their -// adapter; the cast hands them a typed view of the same payload. -export const LLMClient: LLMClient = { prepare: prepare as LLMClient["prepare"], stream, generate } +export const LLMClient = { + Service, + layer, +} as const diff --git a/packages/llm/src/adapter/index.ts b/packages/llm/src/adapter/index.ts index 54d4c448c4..c64546adfc 100644 --- a/packages/llm/src/adapter/index.ts +++ b/packages/llm/src/adapter/index.ts @@ -7,7 +7,8 @@ export type { AdapterRoutedModelInput, AnyAdapter, HttpContext, - LLMClient as LLMClientShape, + Interface as LLMClientShape, + Service as LLMClientService, ModelCapabilitiesInput, ModelRefInput, } from "./client" diff --git a/packages/llm/src/index.ts b/packages/llm/src/index.ts index 3971a34a0d..c610e71a5c 100644 --- a/packages/llm/src/index.ts +++ b/packages/llm/src/index.ts @@ -2,7 +2,8 @@ export { LLMClient, modelCapabilities, modelLimits, modelRef } from "./adapter/c export type { AdapterModelInput, AdapterRoutedModelInput, - LLMClient as LLMClientShape, + Interface as LLMClientShape, + Service as LLMClientService, ModelCapabilitiesInput, ModelRefInput, } from "./adapter/client" diff --git a/packages/llm/src/llm.ts b/packages/llm/src/llm.ts index 3efbcba1da..0ad4df60f4 100644 --- a/packages/llm/src/llm.ts +++ b/packages/llm/src/llm.ts @@ -1,15 +1,10 @@ -import { Context, Effect, Layer, Stream } from "effect" import { - LLMClient, modelCapabilities, modelLimits, modelRef, type ModelCapabilitiesInput, type ModelRefInput, } from "./adapter/client" -import { RequestExecutor } from "./adapter/executor" -import { type Tools } from "./tool" -import { ToolRuntime, type RunOptions } from "./tool-runtime" import { GenerationOptions, HttpOptions, @@ -23,64 +18,7 @@ import { type SystemPart, ToolCallPart, ToolResultPart, - mergeGenerationOptions, - mergeHttpOptions, - mergeProviderOptions, } from "./schema" -import type { LLMError } from "./schema" - -export type StreamWithToolsInput = Omit & Omit, "request"> - -export interface Interface { - readonly stream: (input: LLMRequest | RequestInput) => Stream.Stream - readonly generate: (input: LLMRequest | RequestInput) => Effect.Effect - readonly streamWithTools: ( - input: StreamWithToolsInput, - ) => Stream.Stream -} - -export class Service extends Context.Service()("@opencode/LLM") {} - -const requestOf = (input: LLMRequest | RequestInput) => (input instanceof LLMRequest ? input : request(input)) - -export const make = (executor: RequestExecutor.Interface): Interface => ({ - stream: (input) => - LLMClient.stream(requestOf(input)).pipe(Stream.provideService(RequestExecutor.Service, executor)), - generate: (input) => - LLMClient.generate(requestOf(input)).pipe(Effect.provideService(RequestExecutor.Service, executor)), - streamWithTools: (input) => { - const { maxSteps, concurrency, stopWhen, tools, ...rest } = input - return ToolRuntime.run({ request: request(rest), tools, maxSteps, concurrency, stopWhen }).pipe( - Stream.provideService(RequestExecutor.Service, executor), - ) - }, -}) - -export const layer: Layer.Layer = Layer.effect( - Service, - Effect.gen(function* () { - return Service.of(make(yield* RequestExecutor.Service)) - }), -) - -export const stream = (input: LLMRequest | RequestInput) => - Stream.unwrap( - Effect.gen(function* () { - return (yield* Service).stream(input) - }), - ) - -export const generate = (input: LLMRequest | RequestInput) => - Effect.gen(function* () { - return yield* (yield* Service).generate(input) - }) - -export const streamWithTools = (input: StreamWithToolsInput) => - Stream.unwrap( - Effect.gen(function* () { - return (yield* Service).streamWithTools(input) - }), - ) export type CapabilitiesInput = ModelCapabilitiesInput @@ -95,7 +33,7 @@ export type ToolResultInput = Parameters[0] export type RequestInput = Omit< ConstructorParameters[0], - "system" | "messages" | "tools" | "toolChoice" | "generation" | "http" + "system" | "messages" | "tools" | "toolChoice" | "generation" | "http" | "providerOptions" > & { readonly system?: string | SystemPart | ReadonlyArray readonly prompt?: string | ContentPart | ReadonlyArray @@ -103,6 +41,7 @@ export type RequestInput = Omit< readonly tools?: ReadonlyArray[0]> readonly toolChoice?: ToolChoiceInput readonly generation?: GenerationOptions | ConstructorParameters[0] + readonly providerOptions?: ConstructorParameters[0]["providerOptions"] readonly http?: HttpOptions | ConstructorParameters[0] } @@ -183,9 +122,9 @@ export const request = (input: RequestInput) => { messages: [...(messages?.map(message) ?? []), ...(prompt === undefined ? [] : [user(prompt)])], tools: tools?.map(toolDefinition) ?? [], toolChoice: requestToolChoice ? toolChoice(requestToolChoice) : undefined, - generation: mergeGenerationOptions(input.model.generation, generation(requestGeneration)) ?? generation(), - providerOptions: mergeProviderOptions(input.model.providerOptions, requestProviderOptions), - http: mergeHttpOptions(input.model.http, http(requestHttp)), + generation: requestGeneration === undefined ? undefined : generation(requestGeneration), + providerOptions: requestProviderOptions, + http: http(requestHttp), }) } diff --git a/packages/llm/src/protocols/anthropic-messages.ts b/packages/llm/src/protocols/anthropic-messages.ts index 8040ab904b..2f6ca16a51 100644 --- a/packages/llm/src/protocols/anthropic-messages.ts +++ b/packages/llm/src/protocols/anthropic-messages.ts @@ -314,6 +314,7 @@ const lowerThinking = Effect.fn("AnthropicMessages.lowerThinking")(function* (re const toPayload = Effect.fn("AnthropicMessages.toPayload")(function* (request: LLMRequest) { const toolChoice = request.toolChoice ? yield* lowerToolChoice(request.toolChoice) : undefined + const generation = request.generation return { model: request.model.id, system: request.system.length === 0 @@ -323,11 +324,11 @@ const toPayload = Effect.fn("AnthropicMessages.toPayload")(function* (request: L tools: request.tools.length === 0 || request.toolChoice?.type === "none" ? undefined : request.tools.map(lowerTool), tool_choice: toolChoice, stream: true as const, - max_tokens: request.generation.maxTokens ?? request.model.limits.output ?? 4096, - temperature: request.generation.temperature, - top_p: request.generation.topP, - top_k: request.generation.topK, - stop_sequences: request.generation.stop, + max_tokens: generation?.maxTokens ?? request.model.limits.output ?? 4096, + temperature: generation?.temperature, + top_p: generation?.topP, + top_k: generation?.topK, + stop_sequences: generation?.stop, thinking: yield* lowerThinking(request), } }) diff --git a/packages/llm/src/protocols/bedrock-converse.ts b/packages/llm/src/protocols/bedrock-converse.ts index dea777c86a..bdb5544985 100644 --- a/packages/llm/src/protocols/bedrock-converse.ts +++ b/packages/llm/src/protocols/bedrock-converse.ts @@ -326,21 +326,22 @@ const lowerSystem = (system: ReadonlyArray): Bedro const toPayload = Effect.fn("BedrockConverse.toPayload")(function* (request: LLMRequest) { const toolChoice = request.toolChoice ? yield* lowerToolChoice(request.toolChoice) : undefined + const generation = request.generation return { modelId: request.model.id, messages: yield* lowerMessages(request), system: request.system.length === 0 ? undefined : lowerSystem(request.system), inferenceConfig: - request.generation.maxTokens === undefined && - request.generation.temperature === undefined && - request.generation.topP === undefined && - (request.generation.stop === undefined || request.generation.stop.length === 0) + generation?.maxTokens === undefined && + generation?.temperature === undefined && + generation?.topP === undefined && + (generation?.stop === undefined || generation.stop.length === 0) ? undefined : { - maxTokens: request.generation.maxTokens, - temperature: request.generation.temperature, - topP: request.generation.topP, - stopSequences: request.generation.stop, + maxTokens: generation?.maxTokens, + temperature: generation?.temperature, + topP: generation?.topP, + stopSequences: generation?.stop, }, toolConfig: request.tools.length > 0 && request.toolChoice?.type !== "none" diff --git a/packages/llm/src/protocols/gemini.ts b/packages/llm/src/protocols/gemini.ts index abc6018d02..a41bc03acf 100644 --- a/packages/llm/src/protocols/gemini.ts +++ b/packages/llm/src/protocols/gemini.ts @@ -265,12 +265,13 @@ const thinkingConfig = (request: LLMRequest) => { const toPayload = Effect.fn("Gemini.toPayload")(function* (request: LLMRequest) { const toolsEnabled = request.tools.length > 0 && request.toolChoice?.type !== "none" + const generation = request.generation const generationConfig = { - maxOutputTokens: request.generation.maxTokens, - temperature: request.generation.temperature, - topP: request.generation.topP, - topK: request.generation.topK, - stopSequences: request.generation.stop, + maxOutputTokens: generation?.maxTokens, + temperature: generation?.temperature, + topP: generation?.topP, + topK: generation?.topK, + stopSequences: generation?.stop, thinkingConfig: thinkingConfig(request), } diff --git a/packages/llm/src/protocols/openai-chat.ts b/packages/llm/src/protocols/openai-chat.ts index b4ab722a88..b28f44fe1e 100644 --- a/packages/llm/src/protocols/openai-chat.ts +++ b/packages/llm/src/protocols/openai-chat.ts @@ -257,6 +257,7 @@ const lowerOptions = Effect.fn("OpenAIChat.lowerOptions")(function* (request: LL const toPayload = Effect.fn("OpenAIChat.toPayload")(function* (request: LLMRequest) { // `toPayload` returns the provider payload only. Endpoint, auth, framing, // validation, and HTTP execution are composed by `Adapter.make`. + const generation = request.generation return { model: request.model.id, messages: yield* lowerMessages(request), @@ -264,13 +265,13 @@ const toPayload = Effect.fn("OpenAIChat.toPayload")(function* (request: LLMReque tool_choice: request.toolChoice ? yield* lowerToolChoice(request.toolChoice) : undefined, stream: true as const, stream_options: { include_usage: true }, - max_tokens: request.generation.maxTokens, - temperature: request.generation.temperature, - top_p: request.generation.topP, - frequency_penalty: request.generation.frequencyPenalty, - presence_penalty: request.generation.presencePenalty, - seed: request.generation.seed, - stop: request.generation.stop, + max_tokens: generation?.maxTokens, + temperature: generation?.temperature, + top_p: generation?.topP, + frequency_penalty: generation?.frequencyPenalty, + presence_penalty: generation?.presencePenalty, + seed: generation?.seed, + stop: generation?.stop, ...(yield* lowerOptions(request)), } }) diff --git a/packages/llm/src/protocols/openai-responses.ts b/packages/llm/src/protocols/openai-responses.ts index e42d0a9e0a..b22ac5a1d1 100644 --- a/packages/llm/src/protocols/openai-responses.ts +++ b/packages/llm/src/protocols/openai-responses.ts @@ -236,15 +236,16 @@ const lowerOptions = Effect.fn("OpenAIResponses.lowerOptions")(function* (reques }) const toPayload = Effect.fn("OpenAIResponses.toPayload")(function* (request: LLMRequest) { + const generation = request.generation return { model: request.model.id, input: yield* lowerMessages(request), tools: request.tools.length === 0 ? undefined : request.tools.map(lowerTool), tool_choice: request.toolChoice ? yield* lowerToolChoice(request.toolChoice) : undefined, stream: true as const, - max_output_tokens: request.generation.maxTokens, - temperature: request.generation.temperature, - top_p: request.generation.topP, + max_output_tokens: generation?.maxTokens, + temperature: generation?.temperature, + top_p: generation?.topP, ...(yield* lowerOptions(request)), } }) diff --git a/packages/llm/src/protocols/shared.ts b/packages/llm/src/protocols/shared.ts index af48a3dcd9..26dd113005 100644 --- a/packages/llm/src/protocols/shared.ts +++ b/packages/llm/src/protocols/shared.ts @@ -1,7 +1,7 @@ import { Buffer } from "node:buffer" import { Cause, Effect, Schema, Stream } from "effect" import * as Sse from "effect/unstable/encoding/Sse" -import { HttpClientRequest, type HttpClientResponse } from "effect/unstable/http" +import { Headers, HttpClientRequest, type HttpClientResponse } from "effect/unstable/http" import { InvalidRequestError, ProviderChunkError, type MediaPart, type ToolResultPart } from "../schema" export const Json = Schema.fromJsonString(Schema.Unknown) @@ -189,10 +189,10 @@ export const validateWith = export const jsonPost = (input: { readonly url: string readonly body: string - readonly headers?: Record + readonly headers?: Headers.Input }) => HttpClientRequest.post(input.url).pipe( - HttpClientRequest.setHeaders({ ...input.headers, "content-type": "application/json" }), + HttpClientRequest.setHeaders(Headers.set(Headers.fromInput(input.headers), "content-type", "application/json")), HttpClientRequest.bodyText(input.body, "application/json"), ) diff --git a/packages/llm/src/protocols/utils/bedrock-auth.ts b/packages/llm/src/protocols/utils/bedrock-auth.ts index d77ed6a082..9688b70f8f 100644 --- a/packages/llm/src/protocols/utils/bedrock-auth.ts +++ b/packages/llm/src/protocols/utils/bedrock-auth.ts @@ -1,5 +1,6 @@ import { AwsV4Signer } from "aws4fetch" import { Effect, Option, Schema } from "effect" +import { Headers } from "effect/unstable/http" import { Auth } from "../../adapter/auth" import type { Auth as AuthFn } from "../../adapter/auth" import type { LLMRequest } from "../../schema" @@ -45,7 +46,7 @@ const credentialsFromInput = (request: LLMRequest): Credentials | undefined => const signRequest = (input: { readonly url: string readonly body: string - readonly headers: Record + readonly headers: Headers.Headers readonly credentials: Credentials }) => Effect.tryPromise({ @@ -83,9 +84,9 @@ export const auth: AuthFn = (input) => { "Bedrock Converse requires either model.apiKey or AWS credentials in model.native.aws_credentials", ) } - const headersForSigning = { ...input.headers, "content-type": "application/json" } + const headersForSigning = Headers.set(input.headers, "content-type", "application/json") const signed = yield* signRequest({ url: input.url, body: input.body, headers: headersForSigning, credentials }) - return { ...headersForSigning, ...signed } + return Headers.setAll(headersForSigning, signed) }) } diff --git a/packages/llm/src/schema.ts b/packages/llm/src/schema.ts index c78a9317bb..b20e200278 100644 --- a/packages/llm/src/schema.ts +++ b/packages/llm/src/schema.ts @@ -382,7 +382,7 @@ export class LLMRequest extends Schema.Class("LLM.Request")({ messages: Schema.Array(Message), tools: Schema.Array(ToolDefinition), toolChoice: Schema.optional(ToolChoice), - generation: GenerationOptions, + generation: Schema.optional(GenerationOptions), providerOptions: Schema.optional(ProviderOptions), http: Schema.optional(HttpOptions), responseFormat: Schema.optional(ResponseFormat), diff --git a/packages/llm/src/tool-runtime.ts b/packages/llm/src/tool-runtime.ts index 8c454bf465..521f774720 100644 --- a/packages/llm/src/tool-runtime.ts +++ b/packages/llm/src/tool-runtime.ts @@ -1,7 +1,6 @@ -import { Effect, Stream } from "effect" +import { Context, Effect, Layer, Stream } from "effect" import type { Concurrency } from "effect/Types" -import { LLMClient } from "./adapter/client" -import type { RequestExecutor } from "./adapter/executor" +import { LLMClient, type Service as LLMClientService } from "./adapter/client" import { type ContentPart, type FinishReason, @@ -44,6 +43,12 @@ export interface RunOptions { readonly stopWhen?: (state: RuntimeState) => boolean } +export interface Interface { + readonly run: (options: RunOptions) => Stream.Stream +} + +export class Service extends Context.Service()("@opencode/LLM/ToolRuntime") {} + /** * Run a model with a typed tool record. The runtime streams the model, on * each `tool-call` event decodes the input against the tool's `parameters` @@ -54,66 +59,73 @@ export interface RunOptions { * `maxSteps` is reached, or when `stopWhen` returns `true`. * * Tool handler dependencies are closed over at tool definition time, so the - * runtime's only environment requirement is the `RequestExecutor.Service`. + * runtime's only environment requirement is the `LLMClient.Service`. */ -export const run = (options: RunOptions): Stream.Stream => { - const maxSteps = options.maxSteps ?? 10 - const concurrency = options.concurrency ?? 10 - const tools = options.tools as Tools - const runtimeTools = toDefinitions(tools) - const runtimeToolNames = new Set(runtimeTools.map((tool) => tool.name)) - const initialRequest = - runtimeTools.length === 0 - ? options.request - : LLMRequest.update(options.request, { - tools: [ - ...options.request.tools.filter((tool) => !runtimeToolNames.has(tool.name)), - ...runtimeTools, - ], - }) - - const loop = (request: LLMRequest, step: number): Stream.Stream => - Stream.unwrap( - Effect.gen(function* () { - const state: StepState = { assistantContent: [], toolCalls: [], finishReason: undefined } - - const modelStream = LLMClient.stream(request).pipe( - Stream.tap((event) => Effect.sync(() => accumulate(state, event))), - ) - - const continuation = Stream.unwrap( - Effect.gen(function* () { - if (state.finishReason !== "tool-calls" || state.toolCalls.length === 0) return Stream.empty - if (options.stopWhen?.({ step, request })) return Stream.empty - if (step + 1 >= maxSteps) return Stream.empty - - const dispatched = yield* Effect.forEach( - state.toolCalls, - (call) => dispatch(tools, call).pipe(Effect.map((result) => [call, result] as const)), - { concurrency }, - ) - const followUp = LLMRequest.update(request, { - messages: [ - ...request.messages, - Message.assistant(state.assistantContent), - ...dispatched.map(([call, result]) => - Message.tool({ id: call.id, name: call.name, result }), - ), +export const layer: Layer.Layer = Layer.effect( + Service, + Effect.gen(function* () { + const client = yield* LLMClient.Service + return Service.of({ + run: (options: RunOptions): Stream.Stream => { + const maxSteps = options.maxSteps ?? 10 + const concurrency = options.concurrency ?? 10 + const tools = options.tools as Tools + const runtimeTools = toDefinitions(tools) + const runtimeToolNames = new Set(runtimeTools.map((tool) => tool.name)) + const initialRequest = runtimeTools.length === 0 + ? options.request + : LLMRequest.update(options.request, { + tools: [ + ...options.request.tools.filter((tool) => !runtimeToolNames.has(tool.name)), + ...runtimeTools, ], }) - return Stream.fromIterable(dispatched.flatMap(([call, result]) => emitEvents(call, result))).pipe( - Stream.concat(loop(followUp, step + 1)), - ) - }), - ) + const loop = (request: LLMRequest, step: number): Stream.Stream => + Stream.unwrap( + Effect.gen(function* () { + const state: StepState = { assistantContent: [], toolCalls: [], finishReason: undefined } - return modelStream.pipe(Stream.concat(continuation)) - }), - ) + const modelStream = client.stream(request).pipe( + Stream.tap((event) => Effect.sync(() => accumulate(state, event))), + ) - return loop(initialRequest, 0) -} + const continuation = Stream.unwrap( + Effect.gen(function* () { + if (state.finishReason !== "tool-calls" || state.toolCalls.length === 0) return Stream.empty + if (options.stopWhen?.({ step, request })) return Stream.empty + if (step + 1 >= maxSteps) return Stream.empty + + const dispatched = yield* Effect.forEach( + state.toolCalls, + (call) => dispatch(tools, call).pipe(Effect.map((result) => [call, result] as const)), + { concurrency }, + ) + const followUp = LLMRequest.update(request, { + messages: [ + ...request.messages, + Message.assistant(state.assistantContent), + ...dispatched.map(([call, result]) => + Message.tool({ id: call.id, name: call.name, result }), + ), + ], + }) + + return Stream.fromIterable(dispatched.flatMap(([call, result]) => emitEvents(call, result))).pipe( + Stream.concat(loop(followUp, step + 1)), + ) + }), + ) + + return modelStream.pipe(Stream.concat(continuation)) + }), + ) + + return loop(initialRequest, 0) + }, + }) + }), +) interface StepState { assistantContent: ContentPart[] @@ -204,4 +216,4 @@ const emitEvents = (call: ToolCallPart, result: ToolResultValue): ReadonlyArray< ] : [{ type: "tool-result", id: call.id, name: call.name, result }] -export * as ToolRuntime from "./tool-runtime" +export const ToolRuntime = { Service, layer } as const diff --git a/packages/llm/test/adapter.test.ts b/packages/llm/test/adapter.test.ts index fa95e6c2e0..877562cc86 100644 --- a/packages/llm/test/adapter.test.ts +++ b/packages/llm/test/adapter.test.ts @@ -101,7 +101,7 @@ const it = testEffect(echoLayer) describe("llm adapter", () => { it.effect("stream and generate use the adapter pipeline", () => Effect.gen(function* () { - const llm = LLMClient + const llm = yield* LLMClient.Service const events = Array.from(yield* llm.stream(request).pipe(Stream.runCollect)) const response = yield* llm.generate(request) @@ -112,7 +112,8 @@ describe("llm adapter", () => { it.effect("selects adapters by request adapter", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const llm = yield* LLMClient.Service + const prepared = yield* llm.prepare( LLM.updateRequest(request, { model: updateModel(request.model, { adapter: "gemini-fake" }) }), ) @@ -122,7 +123,8 @@ describe("llm adapter", () => { it.effect("uses registered adapters by model adapter id", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const llm = yield* LLMClient.Service + const prepared = yield* llm.prepare( LLM.updateRequest(request, { model: updateModel(request.model, { adapter: "gemini-fake" }) }), ) @@ -159,7 +161,8 @@ describe("llm adapter", () => { framing: fakeFraming, }) - const response = yield* LLMClient.generate(request) + const llm = yield* LLMClient.Service + const response = yield* llm.generate(request) expect(response.text).toBe('echo:{"body":"hello"}') }), @@ -167,7 +170,8 @@ describe("llm adapter", () => { it.effect("rejects missing adapter", () => Effect.gen(function* () { - const error = yield* LLMClient + const llm = yield* LLMClient.Service + const error = yield* llm .prepare( LLM.updateRequest(request, { model: updateModel(request.model, { adapter: "missing" }) }), ) diff --git a/packages/llm/test/exports.test.ts b/packages/llm/test/exports.test.ts index 6e2d5e9b55..45a57c8a6b 100644 --- a/packages/llm/test/exports.test.ts +++ b/packages/llm/test/exports.test.ts @@ -8,8 +8,9 @@ import * as AnthropicMessages from "@opencode-ai/llm/protocols/anthropic-message describe("public exports", () => { test("root exposes app-facing runtime APIs", () => { - expect(LLM.generate).toBeFunction() - expect(LLMClient.generate).toBeFunction() + expect(LLM.request).toBeFunction() + expect(LLMClient.Service).toBeFunction() + expect(LLMClient.layer).toBeDefined() }) test("adapter barrel exposes adapter-authoring APIs", () => { diff --git a/packages/llm/test/lib/http.ts b/packages/llm/test/lib/http.ts index c7f40f8511..09bbed54b3 100644 --- a/packages/llm/test/lib/http.ts +++ b/packages/llm/test/lib/http.ts @@ -1,6 +1,10 @@ import { Effect, Layer, Ref } from "effect" import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http" -import { RequestExecutor } from "../../src/adapter" +import { LLMClient, RequestExecutor } from "../../src/adapter" +import type { Service as LLMClientService } from "../../src/adapter/client" +import type { Service as RequestExecutorService } from "../../src/adapter/executor" +import { ToolRuntime } from "../../src/tool-runtime" +import type { Service as ToolRuntimeService } from "../../src/tool-runtime" export type HandlerInput = { readonly request: HttpClientRequest.HttpClientRequest @@ -26,8 +30,13 @@ const handlerLayer = (handler: Handler): Layer.Layer => ), ) -const executorWith = (layer: Layer.Layer) => - RequestExecutor.layer.pipe(Layer.provide(layer)) +export type RuntimeEnv = RequestExecutorService | LLMClientService | ToolRuntimeService + +export const runtimeLayer = (layer: Layer.Layer): Layer.Layer => { + const requestExecutorLayer = RequestExecutor.layer.pipe(Layer.provide(layer)) + const llmClientLayer = LLMClient.layer.pipe(Layer.provide(requestExecutorLayer)) + return Layer.mergeAll(requestExecutorLayer, llmClientLayer, ToolRuntime.layer.pipe(Layer.provide(llmClientLayer))) +} const SSE_HEADERS = { "content-type": "text/event-stream" } as const @@ -40,12 +49,12 @@ const SSE_HEADERS = { "content-type": "text/event-stream" } as const export const fixedResponse = ( body: ConstructorParameters[0], init: ResponseInit = { headers: SSE_HEADERS }, -) => executorWith(handlerLayer((input) => Effect.succeed(input.respond(body, init)))) +) => runtimeLayer(handlerLayer((input) => Effect.succeed(input.respond(body, init)))) /** * Layer that builds a response per request. Useful for echo servers. */ -export const dynamicResponse = (handler: Handler) => executorWith(handlerLayer(handler)) +export const dynamicResponse = (handler: Handler) => runtimeLayer(handlerLayer(handler)) /** * Layer that emits the supplied SSE chunks and then aborts mid-stream. Used to diff --git a/packages/llm/test/lib/llm-client.ts b/packages/llm/test/lib/llm-client.ts new file mode 100644 index 0000000000..de4dbc24b0 --- /dev/null +++ b/packages/llm/test/lib/llm-client.ts @@ -0,0 +1,18 @@ +import { Effect, Layer, Stream } from "effect" +import { LLMClient, RequestExecutor } from "../../src/adapter" +import type { LLMRequest } from "../../src/schema" + +export const prepare = (request: LLMRequest) => + Effect.gen(function* () { + return yield* (yield* LLMClient.Service).prepare(request) + }).pipe(Effect.provide(LLMClient.layer.pipe(Layer.provide(RequestExecutor.defaultLayer)))) + +export const generate = (request: LLMRequest) => + Effect.gen(function* () { + return yield* (yield* LLMClient.Service).generate(request) + }) + +export const stream = (request: LLMRequest) => + Stream.unwrap(Effect.gen(function* () { + return (yield* LLMClient.Service).stream(request) + })) diff --git a/packages/llm/test/lib/tool-runtime.ts b/packages/llm/test/lib/tool-runtime.ts new file mode 100644 index 0000000000..6eebf7f1e5 --- /dev/null +++ b/packages/llm/test/lib/tool-runtime.ts @@ -0,0 +1,8 @@ +import { Effect, Stream } from "effect" +import type { Tools } from "../../src/tool" +import { ToolRuntime, type RunOptions } from "../../src/tool-runtime" + +export const runTools = (options: RunOptions) => + Stream.unwrap(Effect.gen(function* () { + return (yield* ToolRuntime.Service).run(options) + })) diff --git a/packages/llm/test/llm.test.ts b/packages/llm/test/llm.test.ts index 9407c0e7b3..74cd9b0b78 100644 --- a/packages/llm/test/llm.test.ts +++ b/packages/llm/test/llm.test.ts @@ -16,7 +16,7 @@ describe("llm constructors", () => { expect(request.messages[0]).toBeInstanceOf(Message) expect(request.system).toEqual([{ type: "text", text: "You are concise." }]) expect(request.messages[0]?.content).toEqual([{ type: "text", text: "Say hello." }]) - expect(request.generation).toEqual({}) + expect(request.generation).toBeUndefined() expect(request.tools).toEqual([]) }) @@ -38,7 +38,7 @@ describe("llm constructors", () => { expect(updated.messages.map((message) => message.role)).toEqual(["user", "assistant"]) }) - test("merges model defaults with call options", () => { + test("keeps request options separate from model defaults", () => { const request = LLM.request({ model: LLM.model({ id: "fake-model", @@ -54,12 +54,12 @@ describe("llm constructors", () => { http: { body: { metadata: { request: true } }, headers: { "x-shared": "request" }, query: { request: "1" } }, }) - expect(request.generation).toEqual({ maxTokens: 100, temperature: 0 }) - expect(request.providerOptions).toEqual({ openai: { store: true, metadata: { model: true, request: true } } }) + expect(request.generation).toEqual({ temperature: 0 }) + expect(request.providerOptions).toEqual({ openai: { store: true, metadata: { request: true } } }) expect(request.http).toEqual({ - body: { metadata: { model: true, request: true } }, + body: { metadata: { request: true } }, headers: { "x-shared": "request" }, - query: { model: "1", request: "1" }, + query: { request: "1" }, }) }) diff --git a/packages/llm/test/provider/anthropic-messages.recorded.test.ts b/packages/llm/test/provider/anthropic-messages.recorded.test.ts index c41b5be267..624ee78f94 100644 --- a/packages/llm/test/provider/anthropic-messages.recorded.test.ts +++ b/packages/llm/test/provider/anthropic-messages.recorded.test.ts @@ -1,10 +1,11 @@ import { describe, expect } from "bun:test" import { Effect } from "effect" -import { LLM, ProviderRequestError } from "../../src" +import { LLM, ProviderRequestError, type LLMRequest } from "../../src" import { LLMClient } from "../../src/adapter" import * as AnthropicMessages from "../../src/protocols/anthropic-messages" import { eventSummary, expectWeatherToolLoop, runWeatherToolLoop, textRequest, weatherToolLoopRequest, weatherToolName, weatherToolRequest } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" +import * as TestLLMClient from "../lib/llm-client" const model = AnthropicMessages.model({ id: "claude-haiku-4-5-20251001", @@ -31,7 +32,10 @@ const recorded = recordedTests({ requires: ["ANTHROPIC_API_KEY"], options: { requestHeaders: ["content-type", "anthropic-version"] }, }) -const anthropic = LLMClient +const generate = (request: LLMRequest) => + Effect.gen(function* () { + return yield* TestLLMClient.generate(request) + }) const malformedToolOrderRequest = LLM.request({ id: "recorded_anthropic_malformed_tool_order", @@ -50,7 +54,7 @@ const malformedToolOrderRequest = LLM.request({ describe("Anthropic Messages recorded", () => { recorded.effect("streams text", () => Effect.gen(function* () { - const response = yield* anthropic.generate(request) + const response = yield* generate(request) expect(eventSummary(response.events)).toEqual([ { type: "text", value: "Hello!" }, @@ -61,7 +65,7 @@ describe("Anthropic Messages recorded", () => { recorded.effect.with("streams tool call", { tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* anthropic.generate(toolRequest) + const response = yield* generate(toolRequest) expect(eventSummary(response.events)).toEqual([ { type: "tool-call", name: weatherToolName, input: { city: "Paris" } }, @@ -78,7 +82,7 @@ describe("Anthropic Messages recorded", () => { recorded.effect.with("rejects malformed assistant tool order", { tags: ["tool", "sad-path"] }, () => Effect.gen(function* () { - const error = yield* anthropic.generate(malformedToolOrderRequest).pipe(Effect.flip) + const error = yield* generate(malformedToolOrderRequest).pipe(Effect.flip) expect(error).toBeInstanceOf(ProviderRequestError) expect(error).toMatchObject({ status: 400 }) diff --git a/packages/llm/test/provider/anthropic-messages.test.ts b/packages/llm/test/provider/anthropic-messages.test.ts index e6dc5df0b1..2b79753c85 100644 --- a/packages/llm/test/provider/anthropic-messages.test.ts +++ b/packages/llm/test/provider/anthropic-messages.test.ts @@ -1,9 +1,10 @@ import { describe, expect } from "bun:test" -import { Effect, Layer } from "effect" +import { Effect } from "effect" import { CacheHint, LLM, ProviderRequestError } from "../../src" import { LLMClient } from "../../src/adapter" import * as AnthropicMessages from "../../src/protocols/anthropic-messages" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" import { fixedResponse } from "../lib/http" import { sseEvents } from "../lib/sse" @@ -21,12 +22,10 @@ const request = LLM.request({ generation: { maxTokens: 20, temperature: 0 }, }) -const it = testEffect(Layer.empty) - describe("Anthropic Messages adapter", () => { it.effect("prepares Anthropic Messages target", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* TestLLMClient.prepare(request) expect(prepared.payload).toEqual({ model: "claude-sonnet-4-5", @@ -41,7 +40,7 @@ describe("Anthropic Messages adapter", () => { it.effect("prepares tool call and tool result messages", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_tool_result", model, @@ -80,8 +79,7 @@ describe("Anthropic Messages adapter", () => { { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 2 } }, { type: "message_stop" }, ) - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body))) expect(LLM.outputText(response)).toBe("Hello!") @@ -106,8 +104,7 @@ describe("Anthropic Messages adapter", () => { { type: "content_block_stop", index: 0 }, { type: "message_delta", delta: { stop_reason: "tool_use" }, usage: { output_tokens: 1 } }, ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], }), @@ -130,8 +127,7 @@ describe("Anthropic Messages adapter", () => { it.effect("emits provider-error events for mid-stream provider errors", () => Effect.gen(function* () { - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse(sseEvents({ type: "error", error: { type: "overloaded_error", message: "Overloaded" } })), @@ -144,8 +140,7 @@ describe("Anthropic Messages adapter", () => { it.effect("fails HTTP provider errors before stream parsing", () => Effect.gen(function* () { - const error = yield* LLMClient - .generate(request) + const error = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse('{"type":"error","error":{"type":"invalid_request_error","message":"Bad request"}}', { @@ -184,8 +179,7 @@ describe("Anthropic Messages adapter", () => { { type: "content_block_stop", index: 2 }, { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 8 } }, ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "web_search", description: "Web search", inputSchema: { type: "object" } }], }), @@ -232,8 +226,7 @@ describe("Anthropic Messages adapter", () => { { type: "content_block_stop", index: 1 }, { type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 1 } }, ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "web_search", description: "Web search", inputSchema: { type: "object" } }], }), @@ -253,7 +246,7 @@ describe("Anthropic Messages adapter", () => { it.effect("round-trips provider-executed assistant content into server tool blocks", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_round_trip", model, @@ -304,8 +297,7 @@ describe("Anthropic Messages adapter", () => { it.effect("rejects round-trip for unknown server tool names", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_unknown_server_tool", model, @@ -330,8 +322,7 @@ describe("Anthropic Messages adapter", () => { it.effect("rejects unsupported user media content", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_media", model, diff --git a/packages/llm/test/provider/bedrock-converse.test.ts b/packages/llm/test/provider/bedrock-converse.test.ts index 4ddc709f19..da06dd32b3 100644 --- a/packages/llm/test/provider/bedrock-converse.test.ts +++ b/packages/llm/test/provider/bedrock-converse.test.ts @@ -1,11 +1,12 @@ import { EventStreamCodec } from "@smithy/eventstream-codec" import { fromUtf8, toUtf8 } from "@smithy/util-utf8" import { describe, expect } from "bun:test" -import { Effect, Layer } from "effect" +import { Effect } from "effect" import { CacheHint, LLM } from "../../src" import { LLMClient } from "../../src/adapter" import * as BedrockConverse from "../../src/protocols/bedrock-converse" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" import { fixedResponse } from "../lib/http" import { eventSummary, expectWeatherToolLoop, runWeatherToolLoop, weatherTool, weatherToolLoopRequest, weatherToolName } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" @@ -59,12 +60,10 @@ const baseRequest = LLM.request({ generation: { maxTokens: 64, temperature: 0 }, }) -const it = testEffect(Layer.empty) - describe("Bedrock Converse adapter", () => { it.effect("prepares Converse target with system, inference config, and messages", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare(baseRequest) + const prepared = yield* TestLLMClient.prepare(baseRequest) expect(prepared.payload).toEqual({ modelId: "anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -77,7 +76,7 @@ describe("Bedrock Converse adapter", () => { it.effect("prepares tool config with toolSpec and toolChoice", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.updateRequest(baseRequest, { tools: [ { @@ -111,7 +110,7 @@ describe("Bedrock Converse adapter", () => { it.effect("lowers assistant tool-call + tool-result message history", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_history", model, @@ -157,8 +156,7 @@ describe("Bedrock Converse adapter", () => { ["messageStop", { stopReason: "end_turn" }], ["metadata", { usage: { inputTokens: 5, outputTokens: 2, totalTokens: 7 } }], ) - const response = yield* LLMClient - .generate(baseRequest) + const response = yield* TestLLMClient.generate(baseRequest) .pipe(Effect.provide(fixedBytes(body))) expect(LLM.outputText(response)).toBe("Hello!") @@ -192,8 +190,7 @@ describe("Bedrock Converse adapter", () => { ["contentBlockStop", { contentBlockIndex: 0 }], ["messageStop", { stopReason: "tool_use" }], ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(baseRequest, { tools: [{ name: "lookup", description: "Lookup", inputSchema: { type: "object" } }], }), @@ -223,8 +220,7 @@ describe("Bedrock Converse adapter", () => { ["contentBlockStop", { contentBlockIndex: 0 }], ["messageStop", { stopReason: "end_turn" }], ) - const response = yield* LLMClient - .generate(baseRequest) + const response = yield* TestLLMClient.generate(baseRequest) .pipe(Effect.provide(fixedBytes(body))) expect(LLM.outputReasoning(response)).toBe("Let me think.") @@ -237,8 +233,7 @@ describe("Bedrock Converse adapter", () => { ["messageStart", { role: "assistant" }], ["throttlingException", { message: "Slow down" }], ) - const response = yield* LLMClient - .generate(baseRequest) + const response = yield* TestLLMClient.generate(baseRequest) .pipe(Effect.provide(fixedBytes(body))) expect(response.events.find((event) => event.type === "provider-error")).toEqual({ @@ -255,8 +250,7 @@ describe("Bedrock Converse adapter", () => { id: "anthropic.claude-3-5-sonnet-20240620-v1:0", baseURL: "https://bedrock-runtime.test", }) - const error = yield* LLMClient - .generate(LLM.updateRequest(baseRequest, { model: unsignedModel })) + const error = yield* TestLLMClient.generate(LLM.updateRequest(baseRequest, { model: unsignedModel })) .pipe(Effect.provide(fixedBytes(eventStreamBody(["messageStop", { stopReason: "end_turn" }]))), Effect.flip) expect(error.message).toContain("Bedrock Converse requires either model.apiKey") @@ -274,7 +268,7 @@ describe("Bedrock Converse adapter", () => { secretAccessKey: "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", }, }) - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.updateRequest(baseRequest, { model: signed }), ) @@ -291,7 +285,7 @@ describe("Bedrock Converse adapter", () => { it.effect("emits cachePoint markers after system, user-text, and assistant-text with cache hints", () => Effect.gen(function* () { const cache = new CacheHint({ type: "ephemeral" }) - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_cache", model, @@ -323,7 +317,7 @@ describe("Bedrock Converse adapter", () => { it.effect("does not emit cachePoint when no cache hint is set", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare(baseRequest) + const prepared = yield* TestLLMClient.prepare(baseRequest) expect(prepared.payload).toMatchObject({ system: [{ text: "You are concise." }], messages: [{ role: "user", content: [{ text: "Say hello." }] }], @@ -333,7 +327,7 @@ describe("Bedrock Converse adapter", () => { it.effect("lowers image media into Bedrock image blocks", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_image", model, @@ -369,7 +363,7 @@ describe("Bedrock Converse adapter", () => { it.effect("base64-encodes Uint8Array image bytes", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_image_bytes", model, @@ -395,7 +389,7 @@ describe("Bedrock Converse adapter", () => { it.effect("lowers document media into Bedrock document blocks with format and name", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_doc", model, @@ -426,8 +420,7 @@ describe("Bedrock Converse adapter", () => { it.effect("rejects unsupported image media types", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_bad_image", model, @@ -442,8 +435,7 @@ describe("Bedrock Converse adapter", () => { it.effect("rejects unsupported document media types", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_bad_doc", model, @@ -494,7 +486,7 @@ const recorded = recordedTests({ describe("Bedrock Converse recorded", () => { recorded.effect("streams text", () => Effect.gen(function* () { - const llm = LLMClient + const llm = yield* LLMClient.Service const response = yield* llm.generate( LLM.request({ id: "recorded_bedrock_text", @@ -514,7 +506,7 @@ describe("Bedrock Converse recorded", () => { recorded.effect.with("streams a tool call", { tags: ["tool"] }, () => Effect.gen(function* () { - const llm = LLMClient + const llm = yield* LLMClient.Service const response = yield* llm.generate( LLM.request({ id: "recorded_bedrock_tool_call", @@ -536,7 +528,7 @@ describe("Bedrock Converse recorded", () => { recorded.effect.with("drives a tool loop", { tags: ["tool", "tool-loop", "golden"] }, () => Effect.gen(function* () { - const llm = LLMClient + const llm = yield* LLMClient.Service expectWeatherToolLoop(yield* runWeatherToolLoop(weatherToolLoopRequest({ id: "recorded_bedrock_tool_loop", model: recordedModel(), diff --git a/packages/llm/test/provider/gemini.recorded.test.ts b/packages/llm/test/provider/gemini.recorded.test.ts index e56e02395d..a0cbd2801d 100644 --- a/packages/llm/test/provider/gemini.recorded.test.ts +++ b/packages/llm/test/provider/gemini.recorded.test.ts @@ -1,10 +1,11 @@ import { describe, expect } from "bun:test" import { Effect } from "effect" -import { LLM } from "../../src" +import { LLM, type LLMRequest } from "../../src" import { LLMClient } from "../../src/adapter" import * as Gemini from "../../src/protocols/gemini" import { eventSummary, textRequest, weatherToolName, weatherToolRequest } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" +import * as TestLLMClient from "../lib/llm-client" const model = Gemini.model({ id: "gemini-2.5-flash", @@ -20,12 +21,15 @@ const recorded = recordedTests({ protocol: "gemini", requires: ["GOOGLE_GENERATIVE_AI_API_KEY"], }) -const gemini = LLMClient +const generate = (request: LLMRequest) => + Effect.gen(function* () { + return yield* TestLLMClient.generate(request) + }) describe("Gemini recorded", () => { recorded.effect("streams text", () => Effect.gen(function* () { - const response = yield* gemini.generate(request) + const response = yield* generate(request) expect(eventSummary(response.events)).toEqual([ { type: "text", value: expect.stringMatching(/^Hello!?$/) }, @@ -36,7 +40,7 @@ describe("Gemini recorded", () => { recorded.effect.with("streams tool call", { tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* gemini.generate(toolRequest) + const response = yield* generate(toolRequest) expect(eventSummary(response.events)).toEqual([ { type: "tool-call", name: weatherToolName, input: { city: "Paris" } }, diff --git a/packages/llm/test/provider/gemini.test.ts b/packages/llm/test/provider/gemini.test.ts index 8c68d81d66..75d20e9a4c 100644 --- a/packages/llm/test/provider/gemini.test.ts +++ b/packages/llm/test/provider/gemini.test.ts @@ -1,9 +1,10 @@ import { describe, expect } from "bun:test" -import { Effect, Layer } from "effect" +import { Effect } from "effect" import { LLM, ProviderChunkError } from "../../src" import { LLMClient } from "../../src/adapter" import * as Gemini from "../../src/protocols/gemini" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" import { fixedResponse } from "../lib/http" import { sseEvents, sseRaw } from "../lib/sse" @@ -21,12 +22,10 @@ const request = LLM.request({ generation: { maxTokens: 20, temperature: 0 }, }) -const it = testEffect(Layer.empty) - describe("Gemini adapter", () => { it.effect("prepares Gemini target", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* TestLLMClient.prepare(request) expect(prepared.payload).toEqual({ contents: [{ role: "user", parts: [{ text: "Say hello." }] }], @@ -38,7 +37,7 @@ describe("Gemini adapter", () => { it.effect("prepares multimodal user input and tool history", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_tool_result", model, @@ -91,7 +90,7 @@ describe("Gemini adapter", () => { it.effect("omits tools when tool choice is none", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_no_tools", model, @@ -109,7 +108,7 @@ describe("Gemini adapter", () => { it.effect("sanitizes integer enums, dangling required, untyped arrays, and scalar object keys", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_schema_patch", model, @@ -177,8 +176,7 @@ describe("Gemini adapter", () => { }, }, ) - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body))) expect(LLM.outputText(response)).toBe("Hello!") @@ -230,8 +228,7 @@ describe("Gemini adapter", () => { usageMetadata: { promptTokenCount: 5, candidatesTokenCount: 1 }, }, ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], }), @@ -266,8 +263,7 @@ describe("Gemini adapter", () => { }], }, ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], }), @@ -284,15 +280,13 @@ describe("Gemini adapter", () => { it.effect("maps length and content-filter finish reasons", () => Effect.gen(function* () { - const length = yield* LLMClient - .generate(request) + const length = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse(sseEvents({ candidates: [{ content: { role: "model", parts: [] }, finishReason: "MAX_TOKENS" }] })), ), ) - const filtered = yield* LLMClient - .generate(request) + const filtered = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse(sseEvents({ candidates: [{ content: { role: "model", parts: [] }, finishReason: "SAFETY" }] })), @@ -306,8 +300,7 @@ describe("Gemini adapter", () => { it.effect("leaves total usage undefined when component counts are missing", () => Effect.gen(function* () { - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(sseEvents({ usageMetadata: { thoughtsTokenCount: 1 } })))) expect(response.usage).toMatchObject({ reasoningTokens: 1 }) @@ -317,8 +310,7 @@ describe("Gemini adapter", () => { it.effect("fails invalid stream chunks", () => Effect.gen(function* () { - const error = yield* LLMClient - .generate(request) + const error = yield* TestLLMClient.generate(request) .pipe( Effect.provide(fixedResponse(sseRaw("data: {not json}"))), Effect.flip, @@ -331,8 +323,7 @@ describe("Gemini adapter", () => { it.effect("rejects unsupported assistant media content", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_media", model, diff --git a/packages/llm/test/provider/openai-chat-tool-loop.recorded.test.ts b/packages/llm/test/provider/openai-chat-tool-loop.recorded.test.ts index 1213a225af..8833c92c4e 100644 --- a/packages/llm/test/provider/openai-chat-tool-loop.recorded.test.ts +++ b/packages/llm/test/provider/openai-chat-tool-loop.recorded.test.ts @@ -5,6 +5,7 @@ import * as OpenAIChat from "../../src/protocols/openai-chat" import { ToolRuntime } from "../../src/tool-runtime" import { eventSummary, weatherRuntimeTool } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" +import * as TestToolRuntime from "../lib/tool-runtime" // Multi-interaction recorded test: drives the typed `ToolRuntime` against a // live OpenAI Chat endpoint so the cassette captures every model round in @@ -35,7 +36,7 @@ describe("OpenAI Chat tool-loop recorded", () => { recorded.effect.with("drives a tool loop end-to-end", { tags: ["tool", "tool-loop"] }, () => Effect.gen(function* () { const events = Array.from( - yield* ToolRuntime.run({ request, tools: { get_weather: weatherRuntimeTool } }).pipe(Stream.runCollect), + yield* TestToolRuntime.runTools({ request, tools: { get_weather: weatherRuntimeTool } }).pipe(Stream.runCollect), ) expect(LLM.outputText({ events })).toContain("Paris") diff --git a/packages/llm/test/provider/openai-chat.recorded.test.ts b/packages/llm/test/provider/openai-chat.recorded.test.ts index 35cde20c16..07ab49f585 100644 --- a/packages/llm/test/provider/openai-chat.recorded.test.ts +++ b/packages/llm/test/provider/openai-chat.recorded.test.ts @@ -1,10 +1,11 @@ import { describe, expect } from "bun:test" import { Effect } from "effect" -import { LLM } from "../../src" +import { LLM, type LLMRequest } from "../../src" import { LLMClient } from "../../src/adapter" import * as OpenAIChat from "../../src/protocols/openai-chat" import { eventSummary, textRequest, weatherToolName, weatherToolRequest } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" +import * as TestLLMClient from "../lib/llm-client" const model = OpenAIChat.model({ id: "gpt-4o-mini", @@ -36,12 +37,15 @@ const recorded = recordedTests({ protocol: "openai-chat", requires: ["OPENAI_API_KEY"], }) -const openai = LLMClient +const generate = (request: LLMRequest) => + Effect.gen(function* () { + return yield* TestLLMClient.generate(request) + }) describe("OpenAI Chat recorded", () => { recorded.effect("streams text", () => Effect.gen(function* () { - const response = yield* openai.generate(request) + const response = yield* generate(request) expect(eventSummary(response.events)).toEqual([ { type: "text", value: "Hello!" }, @@ -62,7 +66,7 @@ describe("OpenAI Chat recorded", () => { recorded.effect.with("streams tool call", { tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* openai.generate(toolRequest) + const response = yield* generate(toolRequest) expect(eventSummary(response.events)).toEqual([ { type: "tool-call", name: weatherToolName, input: { city: "Paris" } }, @@ -77,7 +81,7 @@ describe("OpenAI Chat recorded", () => { recorded.effect.with("continues after tool result", { tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* openai.generate(toolResultRequest) + const response = yield* generate(toolResultRequest) expect(eventSummary(response.events)).toEqual([ { type: "text", value: "The weather in Paris is sunny with a temperature of 22°C." }, diff --git a/packages/llm/test/provider/openai-chat.test.ts b/packages/llm/test/provider/openai-chat.test.ts index 99c1a77858..8580fb12a7 100644 --- a/packages/llm/test/provider/openai-chat.test.ts +++ b/packages/llm/test/provider/openai-chat.test.ts @@ -1,12 +1,12 @@ import { describe, expect } from "bun:test" -import { Effect, Layer, Schema, Stream } from "effect" +import { Effect, Schema, Stream } from "effect" import { HttpClientRequest } from "effect/unstable/http" import { LLM, ProviderRequestError } from "../../src" -import { LLMClient } from "../../src/adapter" import * as Azure from "../../src/providers/azure" import * as OpenAI from "../../src/providers/openai" import * as OpenAIChat from "../../src/protocols/openai-chat" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" import { dynamicResponse, fixedResponse, truncatedStream } from "../lib/http" import { deltaChunk, usageChunk } from "../lib/openai-chunks" import { sseEvents } from "../lib/sse" @@ -29,15 +29,13 @@ const request = LLM.request({ generation: { maxTokens: 20, temperature: 0 }, }) -const it = testEffect(Layer.empty) - describe("OpenAI Chat adapter", () => { it.effect("prepares OpenAI Chat payload", () => Effect.gen(function* () { // Pass the OpenAIChat payload type so `prepared.payload` is statically // typed to the adapter's native shape — the assertions below read field // names without `unknown` casts. - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* TestLLMClient.prepare(request) const _typed: { readonly model: string; readonly stream: true } = prepared.payload expect(prepared.payload).toEqual({ @@ -56,7 +54,7 @@ describe("OpenAI Chat adapter", () => { it.effect("maps OpenAI provider options to Chat options", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ model: OpenAI.chat("gpt-4o-mini", { baseURL: "https://api.openai.test/v1/" }), prompt: "think", @@ -70,8 +68,7 @@ describe("OpenAI Chat adapter", () => { ) it.effect("adds native query params to the Chat Completions URL", () => - LLMClient - .generate(LLM.updateRequest(request, { model: OpenAIChat.model({ ...model, queryParams: { "api-version": "v1" } }) })) + TestLLMClient.generate(LLM.updateRequest(request, { model: OpenAIChat.model({ ...model, queryParams: { "api-version": "v1" } }) })) .pipe( Effect.provide( dynamicResponse((input) => @@ -88,8 +85,7 @@ describe("OpenAI Chat adapter", () => { ) it.effect("uses Azure api-key header for static OpenAI Chat keys", () => - LLMClient - .generate( + TestLLMClient.generate( LLM.updateRequest(request, { model: Azure.model("gpt-4o-mini", { useCompletionUrls: true, @@ -116,8 +112,7 @@ describe("OpenAI Chat adapter", () => { ) it.effect("applies serializable HTTP overlays after payload lowering", () => - LLMClient - .generate( + TestLLMClient.generate( LLM.updateRequest(request, { model: OpenAIChat.model({ ...model, apiKey: "fresh-key", headers: { authorization: "Bearer stale" } }), http: { @@ -151,7 +146,7 @@ describe("OpenAI Chat adapter", () => { it.effect("prepares assistant tool-call and tool-result messages", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_tool_result", model, @@ -188,8 +183,7 @@ describe("OpenAI Chat adapter", () => { it.effect("rejects unsupported user media content", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_media", model, @@ -204,8 +198,7 @@ describe("OpenAI Chat adapter", () => { it.effect("rejects unsupported assistant reasoning content", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_reasoning", model, @@ -232,8 +225,7 @@ describe("OpenAI Chat adapter", () => { completion_tokens_details: { reasoning_tokens: 0 }, }), ) - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body))) expect(LLM.outputText(response)).toBe("Hello!") @@ -272,8 +264,7 @@ describe("OpenAI Chat adapter", () => { deltaChunk({ tool_calls: [{ index: 0, function: { arguments: ':"weather"}' } }] }), deltaChunk({}, "tool_calls"), ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], }), @@ -298,8 +289,7 @@ describe("OpenAI Chat adapter", () => { }), deltaChunk({ tool_calls: [{ index: 0, function: { arguments: ':"weather"}' } }] }), ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], }), @@ -317,8 +307,7 @@ describe("OpenAI Chat adapter", () => { it.effect("fails on malformed stream chunks", () => Effect.gen(function* () { const body = sseEvents(deltaChunk({ content: 123 })) - const error = yield* LLMClient - .generate(request) + const error = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body)), Effect.flip) expect(error.message).toContain("Invalid openai/openai-chat stream chunk") @@ -330,8 +319,7 @@ describe("OpenAI Chat adapter", () => { const layer = truncatedStream([ `data: ${JSON.stringify(deltaChunk({ role: "assistant", content: "Hello" }))}\n\n`, ]) - const error = yield* LLMClient - .generate(request) + const error = yield* TestLLMClient.generate(request) .pipe(Effect.provide(layer), Effect.flip) expect(error.message).toContain("Failed to read openai/openai-chat stream") @@ -340,8 +328,7 @@ describe("OpenAI Chat adapter", () => { it.effect("fails HTTP provider errors before stream parsing", () => Effect.gen(function* () { - const error = yield* LLMClient - .generate(request) + const error = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse('{"error":{"message":"Bad request","type":"invalid_request_error"}}', { @@ -360,7 +347,6 @@ describe("OpenAI Chat adapter", () => { it.effect("short-circuits the upstream stream when the consumer takes a prefix", () => Effect.gen(function* () { - const llm = LLMClient // The body has more chunks than we'll consume. If `Stream.take(1)` did // not interrupt the upstream HTTP body the test would hang waiting for // the rest of the stream to drain. @@ -371,7 +357,7 @@ describe("OpenAI Chat adapter", () => { ) const events = Array.from( - yield* llm.stream(request).pipe(Stream.take(1), Stream.runCollect, Effect.provide(fixedResponse(body))), + yield* TestLLMClient.stream(request).pipe(Stream.take(1), Stream.runCollect, Effect.provide(fixedResponse(body))), ) expect(events.map((event) => event.type)).toEqual(["text-delta"]) }), diff --git a/packages/llm/test/provider/openai-compatible-chat.recorded.test.ts b/packages/llm/test/provider/openai-compatible-chat.recorded.test.ts index cd6428be81..893a52d7b0 100644 --- a/packages/llm/test/provider/openai-compatible-chat.recorded.test.ts +++ b/packages/llm/test/provider/openai-compatible-chat.recorded.test.ts @@ -1,12 +1,13 @@ import { describe, expect } from "bun:test" import { Effect } from "effect" -import { LLM } from "../../src" +import { LLM, type LLMRequest } from "../../src" import { LLMClient } from "../../src/adapter" import * as OpenAICompatible from "../../src/providers/openai-compatible" import * as OpenAICompatibleChat from "../../src/protocols/openai-compatible-chat" import * as OpenRouter from "../../src/providers/openrouter" import { expectFinish, expectWeatherToolCall, expectWeatherToolLoop, runWeatherToolLoop, textRequest, weatherToolLoopRequest, weatherToolRequest } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" +import * as TestLLMClient from "../lib/llm-client" const deepseekModel = OpenAICompatible.deepseek.model("deepseek-chat", { apiKey: process.env.DEEPSEEK_API_KEY ?? "fixture", @@ -55,7 +56,10 @@ const xaiRequest = textRequest({ id: "recorded_xai_text", model: xaiModel }) const xaiToolRequest = weatherToolRequest({ id: "recorded_xai_tool_call", model: xaiModel }) const recorded = recordedTests({ prefix: "openai-compatible-chat", protocol: "openai-compatible-chat" }) -const llm = LLMClient +const generate = (request: LLMRequest) => + Effect.gen(function* () { + return yield* TestLLMClient.generate(request) + }) const openrouterToolLoops = [ { @@ -81,7 +85,7 @@ const openrouterToolLoops = [ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("deepseek streams text", { provider: "deepseek", requires: ["DEEPSEEK_API_KEY"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(deepseekRequest) + const response = yield* generate(deepseekRequest) expect(LLM.outputText(response)).toMatch(/^Hello!?$/) expectFinish(response.events, "stop") @@ -90,7 +94,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("togetherai streams text", { provider: "togetherai", requires: ["TOGETHER_AI_API_KEY"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(togetherRequest) + const response = yield* generate(togetherRequest) expect(LLM.outputText(response)).toMatch(/^Hello!?$/) expectFinish(response.events, "stop") @@ -99,7 +103,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("togetherai streams tool call", { provider: "togetherai", requires: ["TOGETHER_AI_API_KEY"], tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(togetherToolRequest) + const response = yield* generate(togetherToolRequest) expect(response.events.some((event) => event.type === "tool-input-delta")).toBe(true) expectWeatherToolCall(response) @@ -109,7 +113,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("groq streams text", { provider: "groq", requires: ["GROQ_API_KEY"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(groqRequest) + const response = yield* generate(groqRequest) expect(LLM.outputText(response)).toMatch(/^Hello!?$/) expectFinish(response.events, "stop") @@ -118,7 +122,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("groq streams tool call", { provider: "groq", requires: ["GROQ_API_KEY"], tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(groqToolRequest) + const response = yield* generate(groqToolRequest) expect(response.events.some((event) => event.type === "tool-input-delta")).toBe(true) expectWeatherToolCall(response) @@ -138,7 +142,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("openrouter streams text", { provider: "openrouter", requires: ["OPENROUTER_API_KEY"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(openrouterRequest) + const response = yield* generate(openrouterRequest) expect(LLM.outputText(response)).toMatch(/^Hello!?$/) expectFinish(response.events, "stop") @@ -147,7 +151,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("openrouter streams tool call", { provider: "openrouter", requires: ["OPENROUTER_API_KEY"], tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(openrouterToolRequest) + const response = yield* generate(openrouterToolRequest) expect(response.events.some((event) => event.type === "tool-input-delta")).toBe(true) expectWeatherToolCall(response) @@ -169,7 +173,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("xai streams text", { provider: "xai", requires: ["XAI_API_KEY"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(xaiRequest) + const response = yield* generate(xaiRequest) expect(LLM.outputText(response)).toMatch(/^Hello!?$/) expectFinish(response.events, "stop") @@ -178,7 +182,7 @@ describe("OpenAI-compatible Chat recorded", () => { recorded.effect.with("xai streams tool call", { provider: "xai", requires: ["XAI_API_KEY"], tags: ["tool"] }, () => Effect.gen(function* () { - const response = yield* llm.generate(xaiToolRequest) + const response = yield* generate(xaiToolRequest) expect(response.events.some((event) => event.type === "tool-input-delta")).toBe(true) expectWeatherToolCall(response) diff --git a/packages/llm/test/provider/openai-compatible-chat.test.ts b/packages/llm/test/provider/openai-compatible-chat.test.ts index 613bc17ffe..276883f8fc 100644 --- a/packages/llm/test/provider/openai-compatible-chat.test.ts +++ b/packages/llm/test/provider/openai-compatible-chat.test.ts @@ -1,11 +1,12 @@ import { describe, expect } from "bun:test" -import { Effect, Layer, Schema } from "effect" +import { Effect, Schema } from "effect" import { HttpClientRequest } from "effect/unstable/http" import { LLM } from "../../src" import { LLMClient } from "../../src/adapter" import * as OpenAICompatible from "../../src/providers/openai-compatible" import * as OpenAICompatibleChat from "../../src/protocols/openai-compatible-chat" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" import { dynamicResponse } from "../lib/http" import { sseEvents } from "../lib/sse" @@ -28,8 +29,6 @@ const request = LLM.request({ generation: { maxTokens: 20, temperature: 0 }, }) -const it = testEffect(Layer.empty) - const deltaChunk = (delta: object, finishReason: string | null = null) => ({ id: "chatcmpl_fixture", choices: [{ delta, finish_reason: finishReason }], @@ -54,7 +53,7 @@ const providerFamilies = [ describe("OpenAI-compatible Chat adapter", () => { it.effect("prepares generic Chat target", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], toolChoice: { type: "required" }, @@ -127,7 +126,7 @@ describe("OpenAI-compatible Chat adapter", () => { it.effect("matches AI SDK compatible basic request body fixture", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* TestLLMClient.prepare(request) expect(prepared.payload).toEqual({ model: "deepseek-chat", @@ -145,7 +144,7 @@ describe("OpenAI-compatible Chat adapter", () => { it.effect("matches AI SDK compatible tool request body fixture", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_tool_parity", model, @@ -195,8 +194,7 @@ describe("OpenAI-compatible Chat adapter", () => { it.effect("posts to the configured compatible endpoint and parses text usage", () => Effect.gen(function* () { - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe( Effect.provide( dynamicResponse((input) => diff --git a/packages/llm/test/provider/openai-responses.recorded.test.ts b/packages/llm/test/provider/openai-responses.recorded.test.ts index ee798ee77f..ded13505a5 100644 --- a/packages/llm/test/provider/openai-responses.recorded.test.ts +++ b/packages/llm/test/provider/openai-responses.recorded.test.ts @@ -1,10 +1,11 @@ import { describe, expect } from "bun:test" import { Effect } from "effect" -import { LLM } from "../../src" +import { LLM, type LLMRequest } from "../../src" import { LLMClient } from "../../src/adapter" import * as OpenAIResponses from "../../src/protocols/openai-responses" import { expectFinish, expectWeatherToolCall, expectWeatherToolLoop, runWeatherToolLoop, weatherTool, weatherToolLoopRequest, weatherToolName } from "../recorded-scenarios" import { recordedTests } from "../recorded-test" +import * as TestLLMClient from "../lib/llm-client" const model = OpenAIResponses.model({ id: "gpt-5.5", @@ -41,12 +42,15 @@ const recorded = recordedTests({ protocol: "openai-responses", requires: ["OPENAI_API_KEY"], }) -const openai = LLMClient +const generate = (request: LLMRequest) => + Effect.gen(function* () { + return yield* TestLLMClient.generate(request) + }) describe("OpenAI Responses recorded", () => { recorded.effect.with("gpt-5.5 streams text", { tags: ["flagship"] }, () => Effect.gen(function* () { - const response = yield* openai.generate(textRequest) + const response = yield* generate(textRequest) expect(LLM.outputText(response)).toMatch(/^Hello!?$/) expect(response.usage?.totalTokens).toBeGreaterThan(0) @@ -56,7 +60,7 @@ describe("OpenAI Responses recorded", () => { recorded.effect.with("gpt-5.5 streams tool call", { tags: ["tool", "flagship"] }, () => Effect.gen(function* () { - const response = yield* openai.generate(toolRequest) + const response = yield* generate(toolRequest) expect(response.events.some((event) => event.type === "tool-input-delta")).toBe(true) expect(response.events.find((event) => event.type === "tool-call")).toMatchObject({ diff --git a/packages/llm/test/provider/openai-responses.test.ts b/packages/llm/test/provider/openai-responses.test.ts index da87608ced..d39e9cd120 100644 --- a/packages/llm/test/provider/openai-responses.test.ts +++ b/packages/llm/test/provider/openai-responses.test.ts @@ -1,12 +1,13 @@ import { describe, expect } from "bun:test" -import { Effect, Layer } from "effect" +import { Effect } from "effect" import { HttpClientRequest } from "effect/unstable/http" import { LLM, ProviderRequestError } from "../../src" import { LLMClient } from "../../src/adapter" import * as Azure from "../../src/providers/azure" import * as OpenAI from "../../src/providers/openai" import * as OpenAIResponses from "../../src/protocols/openai-responses" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" import { dynamicResponse, fixedResponse } from "../lib/http" import { sseEvents } from "../lib/sse" @@ -24,12 +25,10 @@ const request = LLM.request({ generation: { maxTokens: 20, temperature: 0 }, }) -const it = testEffect(Layer.empty) - describe("OpenAI Responses adapter", () => { it.effect("prepares OpenAI Responses target", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* TestLLMClient.prepare(request) expect(prepared.payload).toEqual({ model: "gpt-4.1-mini", @@ -46,8 +45,7 @@ describe("OpenAI Responses adapter", () => { it.effect("adds native query params to the Responses URL", () => Effect.gen(function* () { - yield* LLMClient - .generate(LLM.updateRequest(request, { model: OpenAIResponses.model({ ...model, queryParams: { "api-version": "v1" } }) })) + yield* TestLLMClient.generate(LLM.updateRequest(request, { model: OpenAIResponses.model({ ...model, queryParams: { "api-version": "v1" } }) })) .pipe( Effect.provide( dynamicResponse((input) => @@ -66,8 +64,7 @@ describe("OpenAI Responses adapter", () => { it.effect("uses Azure api-key header for static OpenAI Responses keys", () => Effect.gen(function* () { - yield* LLMClient - .generate( + yield* TestLLMClient.generate( LLM.updateRequest(request, { model: Azure.model("gpt-4.1-mini", { baseURL: "https://opencode-test.openai.azure.com/openai/v1/", @@ -95,7 +92,7 @@ describe("OpenAI Responses adapter", () => { it.effect("prepares function call and function output input items", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ id: "req_tool_result", model, @@ -121,7 +118,7 @@ describe("OpenAI Responses adapter", () => { it.effect("maps OpenAI provider options to Responses options", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ model: OpenAI.model("gpt-5.2", { baseURL: "https://api.openai.test/v1/" }), prompt: "think", @@ -146,7 +143,7 @@ describe("OpenAI Responses adapter", () => { it.effect("request OpenAI provider options override model defaults", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ model: OpenAI.model("gpt-4.1-mini", { baseURL: "https://api.openai.test/v1/", @@ -179,8 +176,7 @@ describe("OpenAI Responses adapter", () => { }, }, ) - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body))) expect(LLM.outputText(response)).toBe("Hello!") @@ -230,8 +226,7 @@ describe("OpenAI Responses adapter", () => { }, { type: "response.completed", response: { usage: { input_tokens: 5, output_tokens: 1 } } }, ) - const response = yield* LLMClient - .generate( + const response = yield* TestLLMClient.generate( LLM.updateRequest(request, { tools: [{ name: "lookup", description: "Lookup data", inputSchema: { type: "object" } }], }), @@ -264,8 +259,7 @@ describe("OpenAI Responses adapter", () => { { type: "response.output_item.done", item }, { type: "response.completed", response: { usage: { input_tokens: 5, output_tokens: 1 } } }, ) - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body))) const callsAndResults = response.events.filter((event) => event.type === "tool-call" || event.type === "tool-result") @@ -302,8 +296,7 @@ describe("OpenAI Responses adapter", () => { { type: "response.output_item.done", item }, { type: "response.completed", response: { usage: { input_tokens: 5, output_tokens: 1 } } }, ) - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(body))) const toolCall = response.events.find((event) => event.type === "tool-call") @@ -327,8 +320,7 @@ describe("OpenAI Responses adapter", () => { it.effect("rejects unsupported user media content", () => Effect.gen(function* () { - const error = yield* LLMClient - .prepare( + const error = yield* TestLLMClient.prepare( LLM.request({ id: "req_media", model, @@ -343,8 +335,7 @@ describe("OpenAI Responses adapter", () => { it.effect("emits provider-error events for mid-stream provider errors", () => Effect.gen(function* () { - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse(sseEvents({ type: "error", code: "rate_limit_exceeded", message: "Slow down" })), @@ -357,8 +348,7 @@ describe("OpenAI Responses adapter", () => { it.effect("falls back to error code when no message is present", () => Effect.gen(function* () { - const response = yield* LLMClient - .generate(request) + const response = yield* TestLLMClient.generate(request) .pipe(Effect.provide(fixedResponse(sseEvents({ type: "error", code: "internal_error" })))) expect(response.events).toEqual([{ type: "provider-error", message: "internal_error" }]) @@ -367,8 +357,7 @@ describe("OpenAI Responses adapter", () => { it.effect("fails HTTP provider errors before stream parsing", () => Effect.gen(function* () { - const error = yield* LLMClient - .generate(request) + const error = yield* TestLLMClient.generate(request) .pipe( Effect.provide( fixedResponse('{"error":{"type":"invalid_request_error","message":"Bad request"}}', { diff --git a/packages/llm/test/provider/openrouter.test.ts b/packages/llm/test/provider/openrouter.test.ts index 155944ccdd..a3e246ee7d 100644 --- a/packages/llm/test/provider/openrouter.test.ts +++ b/packages/llm/test/provider/openrouter.test.ts @@ -1,11 +1,10 @@ import { describe, expect } from "bun:test" -import { Effect, Layer } from "effect" +import { Effect } from "effect" import { LLM } from "../../src" import { LLMClient } from "../../src/adapter" import * as OpenRouter from "../../src/providers/openrouter" -import { testEffect } from "../lib/effect" - -const it = testEffect(Layer.empty) +import { it } from "../lib/effect" +import * as TestLLMClient from "../lib/llm-client" describe("OpenRouter", () => { it.effect("prepares OpenRouter models through the OpenAI-compatible Chat route", () => @@ -20,7 +19,7 @@ describe("OpenRouter", () => { apiKey: "test-key", }) - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ model, prompt: "Say hello." }), ) @@ -35,7 +34,7 @@ describe("OpenRouter", () => { it.effect("applies OpenRouter payload options from the model helper", () => Effect.gen(function* () { - const prepared = yield* LLMClient.prepare( + const prepared = yield* TestLLMClient.prepare( LLM.request({ model: OpenRouter.model("anthropic/claude-3.7-sonnet:thinking", { providerOptions: { diff --git a/packages/llm/test/recorded-scenarios.ts b/packages/llm/test/recorded-scenarios.ts index d215ba2fb6..242b653c69 100644 --- a/packages/llm/test/recorded-scenarios.ts +++ b/packages/llm/test/recorded-scenarios.ts @@ -76,10 +76,13 @@ export const weatherToolLoopRequest = (input: { }) export const runWeatherToolLoop = (request: LLMRequest) => - ToolRuntime.run({ request, tools: { [weatherToolName]: weatherRuntimeTool } }).pipe( - Stream.runCollect, - Effect.map((events) => Array.from(events)), - ) + Effect.gen(function* () { + const runtime = yield* ToolRuntime.Service + return yield* runtime.run({ request, tools: { [weatherToolName]: weatherRuntimeTool } }).pipe( + Stream.runCollect, + Effect.map((events) => Array.from(events)), + ) + }) export const expectFinish = ( events: ReadonlyArray, diff --git a/packages/llm/test/recorded-test.ts b/packages/llm/test/recorded-test.ts index e10c9c871b..86227d99c9 100644 --- a/packages/llm/test/recorded-test.ts +++ b/packages/llm/test/recorded-test.ts @@ -1,16 +1,17 @@ import { HttpRecorder } from "@opencode-ai/http-recorder" import { test, type TestOptions } from "bun:test" -import { Effect, Layer } from "effect" +import { Effect } from "effect" import * as fs from "node:fs" import * as path from "node:path" import { fileURLToPath } from "node:url" -import { RequestExecutor } from "../src/adapter" import { testEffect } from "./lib/effect" +import { runtimeLayer, type RuntimeEnv } from "./lib/http" const __dirname = path.dirname(fileURLToPath(import.meta.url)) const FIXTURES_DIR = path.resolve(__dirname, "fixtures", "recordings") type Body = Effect.Effect | (() => Effect.Effect) +type RecordedEnv = RuntimeEnv type RecordedTestsOptions = { readonly prefix: string @@ -107,7 +108,7 @@ export const recordedTests = (options: RecordedTestsOptions) => { const run = ( name: string, caseOptions: RecordedCaseOptions, - body: Body, + body: Body, testOptions?: number | TestOptions, ) => { const cassette = cassetteName(options.prefix, name, caseOptions) @@ -142,21 +143,19 @@ export const recordedTests = (options: RecordedTestsOptions) => { return test.skip(name, () => {}, testOptions) } - return testEffect( - RequestExecutor.layer.pipe(Layer.provide(HttpRecorder.cassetteLayer(cassette, layerOptions))), - ).live(name, body, testOptions) + return testEffect(runtimeLayer(HttpRecorder.cassetteLayer(cassette, layerOptions))).live(name, body, testOptions) } const effect = ( name: string, - body: Body, + body: Body, testOptions?: number | TestOptions, ) => run(name, {}, body, testOptions) effect.with = ( name: string, caseOptions: RecordedCaseOptions, - body: Body, + body: Body, testOptions?: number | TestOptions, ) => run(name, caseOptions, body, testOptions) diff --git a/packages/llm/test/tool-runtime.test.ts b/packages/llm/test/tool-runtime.test.ts index 4fba9ccb7f..dd61fe4034 100644 --- a/packages/llm/test/tool-runtime.test.ts +++ b/packages/llm/test/tool-runtime.test.ts @@ -1,12 +1,13 @@ import { describe, expect } from "bun:test" -import { Effect, Layer, Schema, Stream } from "effect" +import { Effect, Schema, Stream } from "effect" import { LLM, LLMEvent, LLMRequest } from "../src" import { LLMClient } from "../src/adapter" import * as AnthropicMessages from "../src/protocols/anthropic-messages" import * as OpenAIChat from "../src/protocols/openai-chat" import { tool, ToolFailure } from "../src/tool" import { ToolRuntime } from "../src/tool-runtime" -import { testEffect } from "./lib/effect" +import { it } from "./lib/effect" +import * as TestToolRuntime from "./lib/tool-runtime" import { dynamicResponse, scriptedResponses } from "./lib/http" import { deltaChunk, finishChunk, toolCallChunk } from "./lib/openai-chunks" import { sseEvents } from "./lib/sse" @@ -25,8 +26,6 @@ const baseRequest = LLM.request({ prompt: "Use the tool.", }) -const it = testEffect(Layer.empty) - const get_weather = tool({ description: "Get current weather for a city.", parameters: Schema.Struct({ city: Schema.String }), @@ -44,7 +43,7 @@ describe("ToolRuntime", () => { const layer = scriptedResponses([sseEvents(deltaChunk({ role: "assistant", content: "Done." }), finishChunk("stop"))]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -70,7 +69,7 @@ describe("ToolRuntime", () => { }), ) - yield* ToolRuntime.run({ + yield* TestToolRuntime.runTools({ request: LLMRequest.update(baseRequest, { generation: LLM.generation({ maxTokens: 50 }), toolChoice: LLM.toolChoice("auto"), @@ -110,7 +109,7 @@ describe("ToolRuntime", () => { ]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -136,7 +135,7 @@ describe("ToolRuntime", () => { ]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -162,7 +161,7 @@ describe("ToolRuntime", () => { ]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -182,7 +181,7 @@ describe("ToolRuntime", () => { ]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -199,7 +198,7 @@ describe("ToolRuntime", () => { const layer = scriptedResponses([sseEvents(deltaChunk({ role: "assistant", content: "Done." }), finishChunk("stop"))]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -219,7 +218,7 @@ describe("ToolRuntime", () => { const layer = scriptedResponses([toolCallStep, toolCallStep, toolCallStep]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather }, maxSteps: 2 }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather }, maxSteps: 2 }).pipe( Stream.runCollect, Effect.provide(layer), ), @@ -237,7 +236,7 @@ describe("ToolRuntime", () => { ]) const events = Array.from( - yield* ToolRuntime.run({ + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather }, stopWhen: (state) => state.step >= 0, @@ -281,7 +280,7 @@ describe("ToolRuntime", () => { }), ) const events = Array.from( - yield* ToolRuntime.run({ + yield* TestToolRuntime.runTools({ request: LLM.updateRequest(baseRequest, { model: AnthropicMessages.model({ id: "claude-sonnet-4-5", apiKey: "test" }) }), tools: {}, }).pipe( @@ -322,7 +321,7 @@ describe("ToolRuntime", () => { ]) const events = Array.from( - yield* ToolRuntime.run({ request: baseRequest, tools: { get_weather } }).pipe( + yield* TestToolRuntime.runTools({ request: baseRequest, tools: { get_weather } }).pipe( Stream.runCollect, Effect.provide(layer), ), diff --git a/packages/opencode/src/session/llm-native-tools.ts b/packages/opencode/src/session/llm-native-tools.ts index df61227dc1..b15f1abd95 100644 --- a/packages/opencode/src/session/llm-native-tools.ts +++ b/packages/opencode/src/session/llm-native-tools.ts @@ -1,13 +1,12 @@ import { LLM, - LLMClient, type LLMError, type LLMEvent, type LLMRequest, type FinishReason, type ContentPart, + type LLMClientShape, } from "@opencode-ai/llm" -import type { RequestExecutor } from "@opencode-ai/llm/adapter" import { Cause, Deferred, Effect, FiberSet, Queue, Stream, type Scope } from "effect" import type { Tool, ToolExecutionOptions } from "ai" @@ -129,6 +128,7 @@ const dispatchTool = ( // `done` resolves with the accumulated state so the multi-round driver can // decide whether to recurse. const runOneRound = ( + client: LLMClientShape, request: LLMRequest, tools: Record, abort: AbortSignal, @@ -138,7 +138,7 @@ const runOneRound = ( readonly done: Deferred.Deferred }, never, - Scope.Scope | RequestExecutor.Service + Scope.Scope > => Effect.gen(function* () { const queue = yield* Queue.unbounded() @@ -148,7 +148,7 @@ const runOneRound = ( yield* Effect.forkScoped( Effect.gen(function* () { - yield* LLMClient.stream(request).pipe( + yield* client.stream(request).pipe( Stream.runForEach((event) => Effect.gen(function* () { accumulate(state, event) @@ -218,16 +218,17 @@ const continuationRequest = (request: LLMRequest, state: RoundState): LLMRequest * interrupted (e.g. via the abort signal). */ export const runWithTools = (input: { + readonly client: LLMClientShape readonly request: LLMRequest readonly tools: Record readonly abort: AbortSignal readonly maxSteps?: number -}): Stream.Stream => { +}): Stream.Stream => { const maxSteps = input.maxSteps ?? DEFAULT_MAX_STEPS - const round = (request: LLMRequest, step: number): Stream.Stream => + const round = (request: LLMRequest, step: number): Stream.Stream => Stream.unwrap( Effect.gen(function* () { - const { events, done } = yield* runOneRound(request, input.tools, input.abort) + const { events, done } = yield* runOneRound(input.client, request, input.tools, input.abort) const continuation = Stream.unwrap( Effect.gen(function* () { const state = yield* Deferred.await(done) diff --git a/packages/opencode/src/session/llm.ts b/packages/opencode/src/session/llm.ts index 1becded501..fd4939a546 100644 --- a/packages/opencode/src/session/llm.ts +++ b/packages/opencode/src/session/llm.ts @@ -8,6 +8,7 @@ import { mergeDeep } from "remeda" import { GitLabWorkflowLanguageModel } from "gitlab-ai-provider" import { LLMClient, + type LLMClientService, type ProtocolID, } from "@opencode-ai/llm" import { RequestExecutor } from "@opencode-ai/llm/adapter" @@ -103,7 +104,7 @@ const live: Layer.Layer< | Provider.Service | Plugin.Service | Permission.Service - | RequestExecutor.Service + | LLMClientService > = Layer.effect( Service, Effect.gen(function* () { @@ -112,11 +113,7 @@ const live: Layer.Layer< const provider = yield* Provider.Service const plugin = yield* Plugin.Service const perm = yield* Permission.Service - // Required by the LLM-native stream path. The default layer wires it on - // top of `FetchHttpClient.layer`. Yielded here (not inside `runNative`) - // so the executor instance is shared across every native stream the - // service hands out. - const executor = yield* RequestExecutor.Service + const llmClient = yield* LLMClient.Service const prepare = Effect.fn("LLM.prepareStream")(function* (input: StreamRequest) { const [language, cfg, item, info] = yield* Effect.all( @@ -581,14 +578,14 @@ const live: Layer.Layer< const upstream = filteredNativeTools && filteredNativeTools.length > 0 ? LLMNativeTools.runWithTools({ request: llmRequest, + client: llmClient, tools: filteredAITools, abort: input.abort, }) - : LLMClient.stream(llmRequest) + : llmClient.stream(llmRequest) return upstream.pipe( Stream.flatMap((event) => Stream.fromIterable(map.map(event))), Stream.concat(Stream.unwrap(Effect.sync(() => Stream.fromIterable(map.flush())))), - Stream.provideService(RequestExecutor.Service, executor), ) }) @@ -620,15 +617,16 @@ const live: Layer.Layer< export const layer = live.pipe(Layer.provide(Permission.defaultLayer)) -export const defaultLayer = Layer.suspend(() => - layer.pipe( +export const defaultLayer = Layer.suspend(() => { + const llmClientLayer = LLMClient.layer.pipe(Layer.provide(RequestExecutor.defaultLayer)) + return layer.pipe( Layer.provide(Auth.defaultLayer), Layer.provide(Config.defaultLayer), Layer.provide(Provider.defaultLayer), Layer.provide(Plugin.defaultLayer), - Layer.provide(RequestExecutor.defaultLayer), - ), -) + Layer.provide(llmClientLayer), + ) +}) function resolveTools(input: Pick) { const disabled = Permission.disabled( diff --git a/packages/opencode/test/session/llm-native-stream.test.ts b/packages/opencode/test/session/llm-native-stream.test.ts index 132c44c4b1..fc0fdd47b1 100644 --- a/packages/opencode/test/session/llm-native-stream.test.ts +++ b/packages/opencode/test/session/llm-native-stream.test.ts @@ -11,7 +11,7 @@ import { LLMNative } from "../../src/session/llm-native" import { LLMNativeEvents } from "../../src/session/llm-native-events" import { LLMNativeTools } from "../../src/session/llm-native-tools" import { ProviderTest } from "../fake/provider" -import { testEffect } from "../lib/effect" +import { it } from "../lib/effect" import type { MessageV2 } from "../../src/session/message-v2" import type { Provider } from "../../src/provider/provider" import type { Tool } from "../../src/tool/tool" @@ -19,8 +19,8 @@ import type { Tool } from "../../src/tool/tool" // Inline HTTP layer that returns a single fixed body. Mirrors the // `fixedResponse` helper in `packages/llm/test/lib/http.ts` — duplicated here // rather than imported across packages so this test stays self-contained. -const fixedResponse = (body: BodyInit, init: ResponseInit = { headers: { "content-type": "text/event-stream" } }) => - RequestExecutor.layer.pipe( +const fixedResponse = (body: BodyInit, init: ResponseInit = { headers: { "content-type": "text/event-stream" } }) => { + const requestExecutorLayer = RequestExecutor.layer.pipe( Layer.provide( Layer.succeed( HttpClient.HttpClient, @@ -30,12 +30,14 @@ const fixedResponse = (body: BodyInit, init: ResponseInit = { headers: { "conten ), ), ) + return Layer.merge(requestExecutorLayer, LLMClient.layer.pipe(Layer.provide(requestExecutorLayer))) +} // Scripted multi-response HTTP layer. Each request consumes the next body in // order; the final body repeats if more requests arrive. Mirrors the // `scriptedResponses` helper in `packages/llm/test/lib/http.ts`. -const scriptedResponses = (bodies: ReadonlyArray, init: ResponseInit = { headers: { "content-type": "text/event-stream" } }) => - RequestExecutor.layer.pipe( +const scriptedResponses = (bodies: ReadonlyArray, init: ResponseInit = { headers: { "content-type": "text/event-stream" } }) => { + const requestExecutorLayer = RequestExecutor.layer.pipe( Layer.provide( Layer.unwrap( Effect.gen(function* () { @@ -54,6 +56,8 @@ const scriptedResponses = (bodies: ReadonlyArray, init: ResponseInit = ), ), ) + return Layer.merge(requestExecutorLayer, LLMClient.layer.pipe(Layer.provide(requestExecutorLayer))) +} // Encode an Anthropic SSE body. Each event becomes a `data:` line; the codec // also expects `event:` lines but the package's SSE framing only reads the @@ -91,8 +95,6 @@ const userMessage = (mdl: Provider.Model, id: MessageID, parts: MessageV2.Part[] parts, }) -const it = testEffect(Layer.empty) - describe("LLMNative stream wire-up (audit gap #4 phase 1)", () => { it.effect("converts an Anthropic SSE response into session events via the LLMNative path", () => Effect.gen(function* () { @@ -120,7 +122,9 @@ describe("LLMNative stream wire-up (audit gap #4 phase 1)", () => { { type: "message_stop" }, ]) - const events = yield* LLMClient.stream(llmRequest).pipe( + const events = yield* Stream.unwrap(Effect.gen(function* () { + return (yield* LLMClient.Service).stream(llmRequest) + })).pipe( Stream.flatMap((event) => Stream.fromIterable(map.map(event))), Stream.concat(Stream.unwrap(Effect.sync(() => Stream.fromIterable(map.flush())))), Stream.runCollect, @@ -226,12 +230,14 @@ describe("LLMNative stream wire-up (audit gap #4 phase 1)", () => { ]) const map = LLMNativeEvents.mapper() - - const events = yield* LLMNativeTools.runWithTools({ - request: llmRequest, - tools: { lookup: aiTool }, - abort: new AbortController().signal, - }).pipe( + const events = yield* Stream.unwrap(Effect.gen(function* () { + return LLMNativeTools.runWithTools({ + client: yield* LLMClient.Service, + request: llmRequest, + tools: { lookup: aiTool }, + abort: new AbortController().signal, + }) + })).pipe( Stream.flatMap((event) => Stream.fromIterable(map.map(event))), Stream.concat(Stream.unwrap(Effect.sync(() => Stream.fromIterable(map.flush())))), Stream.runCollect, @@ -300,7 +306,9 @@ describe("LLMNative stream wire-up (audit gap #4 phase 1)", () => { tools: [lookupTool], }) - const prepared = yield* LLMClient.prepare(llmRequest) + const prepared = yield* Effect.gen(function* () { + return yield* (yield* LLMClient.Service).prepare(llmRequest) + }).pipe(Effect.provide(LLMClient.layer.pipe(Layer.provide(RequestExecutor.defaultLayer)))) expect(prepared.payload).toMatchObject({ tools: [ { diff --git a/packages/opencode/test/session/llm-native.test.ts b/packages/opencode/test/session/llm-native.test.ts index b723b21620..7b2f663159 100644 --- a/packages/opencode/test/session/llm-native.test.ts +++ b/packages/opencode/test/session/llm-native.test.ts @@ -1,7 +1,8 @@ import { describe, expect } from "bun:test" -import { LLMClient } from "@opencode-ai/llm" +import { LLMClient, type LLMRequest } from "@opencode-ai/llm" +import { RequestExecutor } from "@opencode-ai/llm/adapter" import "@opencode-ai/llm/protocols" -import { Cause, Effect, Exit, Layer, Schema } from "effect" +import { Cause, Effect, Layer, Exit, Schema } from "effect" import { ModelID, ProviderID } from "../../src/provider/schema" import { LLMNative } from "../../src/session/llm-native" import { MessageID, PartID, SessionID } from "../../src/session/schema" @@ -113,7 +114,12 @@ const lookupTool = { execute: () => Effect.succeed({ title: "", metadata: {}, output: "" }), } satisfies Tool.Def -const it = testEffect(Layer.empty) +const prepare = (request: LLMRequest) => + Effect.gen(function* () { + return yield* (yield* LLMClient.Service).prepare(request) + }) + +const it = testEffect(LLMClient.layer.pipe(Layer.provide(RequestExecutor.defaultLayer))) const isRecord = (value: unknown): value is Record => typeof value === "object" && value !== null && !Array.isArray(value) @@ -598,7 +604,7 @@ describe("LLMNative.request", () => { tools: [lookupTool], toolChoice: "lookup", }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(prepared.payload).toMatchObject({ model: "gpt-5", @@ -657,7 +663,7 @@ describe("LLMNative.request", () => { tools: [lookupTool], toolChoice: "lookup", }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(request.model).toMatchObject({ provider: "anthropic", @@ -726,7 +732,7 @@ describe("LLMNative.request", () => { tools: [lookupTool], toolChoice: "lookup", }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(request.model).toMatchObject({ provider: "togetherai", @@ -857,7 +863,7 @@ describe("LLMNative.request", () => { tools: [lookupTool], toolChoice: "lookup", }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(request.model).toMatchObject({ provider: "google", @@ -929,7 +935,7 @@ describe("LLMNative.request", () => { system: ["First", "Second", "Third"], messages: [userMessage(mdl, userID, [textPart(userID, "hello")])], }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(prepared.payload).toMatchObject({ system: [ @@ -951,7 +957,7 @@ describe("LLMNative.request", () => { model: mdl, messages: messageIds.map((id, index) => userMessage(mdl, id, [textPart(id, `m${index}`)])), }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(prepared.payload).toMatchObject({ messages: [ @@ -975,7 +981,7 @@ describe("LLMNative.request", () => { system: ["You are concise."], messages: [userMessage(mdl, userID, [textPart(userID, "hello")])], }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(prepared.payload).toMatchObject({ system: [{ text: "You are concise." }, { cachePoint: { type: "default" } }], @@ -1000,7 +1006,7 @@ describe("LLMNative.request", () => { system: ["A", "B", "C"], messages: ids.map((id, index) => userMessage(mdl, id, [textPart(id, `m${index}`)])), }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) // The serialized OpenAI Responses payload has no cache concept; the // assertion is that nothing in the payload carries a cache marker. @@ -1076,7 +1082,7 @@ describe("LLMNative.request", () => { ]), ], }) - const prepared = yield* LLMClient.prepare(request) + const prepared = yield* prepare(request) expect(prepared.payload).toMatchObject({ messages: [