Browse Source

feat(ai): support image-guided generation (#37998)

Co-authored-by: Aiden Cline <rekram1-node@users.noreply.github.com>
opencode-agent[bot] 3 weeks ago
parent
commit
e93fa06d05

+ 43 - 1
packages/ai/README.md

@@ -29,7 +29,7 @@ Run `LLMClient.stream(request)` instead of `generate` when you want incremental
 Use `Image.generate` with an image model for direct asset generation:
 
 ```ts
-import { Image } from "@opencode-ai/ai"
+import { Image, ImageInput } from "@opencode-ai/ai"
 import { OpenAI } from "@opencode-ai/ai/providers"
 
 const program = Effect.gen(function* () {
@@ -49,6 +49,48 @@ const program = Effect.gen(function* () {
 })
 ```
 
+Pass ordered image inputs to the same method for editing, composition, or image-conditioned generation:
+
+```ts
+const response =
+  yield *
+  Image.generate({
+    model,
+    prompt: "Combine these product photos into one studio scene",
+    images: [
+      ImageInput.bytes(firstBytes, "image/png"),
+      ImageInput.url("https://example.com/second.webp"),
+      ImageInput.file("file_123"),
+    ],
+    options,
+    http,
+  })
+```
+
+`ImageInput.fileUri(uri, mediaType)` represents provider file URIs such as Gemini Files. Raw strings are not
+accepted as image inputs, avoiding ambiguity between base64, URLs, and provider IDs. Empty or omitted `images`
+uses text-to-image generation; a non-empty array selects the provider's edit behavior without enforcing provider
+image-count limits locally. `images` is the only common image-editing field. OpenAI uses multipart for byte/data-URL
+edits and its JSON reference body for URL or file-ID edits. Its provider-specific `options.mask` accepts an
+`ImageInput` for inpainting:
+
+```ts
+yield *
+  Image.generate({
+    model: OpenAI.configure({ apiKey }).image("gpt-image-2"),
+    prompt,
+    images: [ImageInput.bytes(sourceBytes, "image/png")],
+    options: { mask: ImageInput.bytes(maskBytes, "image/png") },
+  })
+```
+
+The OpenAI adapter extracts this helper value into the edit request's native `mask` field rather than passing the
+tagged `ImageInput` object through as an ordinary option. On multipart requests, `http.body` can override option
+fields but not structural `model`, `prompt`, `image[]`, or `mask` fields, and the transport owns the multipart
+`Content-Type` boundary. For JSON requests, `http.body` remains the final raw-native overlay. Gemini does not fetch
+public HTTP URLs, and hosted Z.ai image generation does not accept image inputs. These cases fail with
+`InvalidRequest` before network I/O.
+
 Provider-native image options belong to each request. Raw `http.body` fields have final precedence over them:
 
 ```ts

+ 35 - 0
packages/ai/src/image.ts

@@ -55,9 +55,44 @@ export const ImageModelSchema = Schema.declare((value): value is ImageModel => v
   expected: "Image.Model",
 })
 
+const ImageBytesInput = Schema.Struct({
+  type: Schema.Literal("bytes"),
+  data: Schema.Uint8Array,
+  mediaType: Schema.String,
+})
+const ImageUrlInput = Schema.Struct({
+  type: Schema.Literal("url"),
+  url: Schema.String,
+})
+const ImageFileIDInput = Schema.Struct({
+  type: Schema.Literal("file-id"),
+  id: Schema.String,
+})
+const ImageFileURIInput = Schema.Struct({
+  type: Schema.Literal("file-uri"),
+  uri: Schema.String,
+  mediaType: Schema.String,
+})
+
+export const ImageInputSchema = Schema.Union([
+  ImageBytesInput,
+  ImageUrlInput,
+  ImageFileIDInput,
+  ImageFileURIInput,
+]).pipe(Schema.toTaggedUnion("type"))
+export type ImageInput = Schema.Schema.Type<typeof ImageInputSchema>
+
+export const ImageInput = {
+  bytes: (data: Uint8Array, mediaType: string): ImageInput => ({ type: "bytes", data, mediaType }),
+  url: (url: string): ImageInput => ({ type: "url", url }),
+  file: (id: string): ImageInput => ({ type: "file-id", id }),
+  fileUri: (uri: string, mediaType: string): ImageInput => ({ type: "file-uri", uri, mediaType }),
+} as const
+
 export class ImageRequest extends Schema.Class<ImageRequest>("Image.Request")({
   model: ImageModelSchema,
   prompt: Schema.String,
+  images: Schema.optional(Schema.Array(ImageInputSchema)),
   options: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
   http: Schema.optional(HttpOptions),
 }) {

+ 1 - 1
packages/ai/src/index.ts

@@ -11,7 +11,7 @@ export type {
   Service as LLMClientService,
 } from "./route/client"
 export * from "./schema"
-export { GeneratedImage, ImageModel, ImageRequest, ImageResponse } from "./image"
+export { GeneratedImage, ImageInput, ImageInputSchema, ImageModel, ImageRequest, ImageResponse } from "./image"
 export type { ImageModelOptions, ImageOptions, ImageRequestFor, ImageRequestInput, ImageRoute } from "./image"
 export { Image } from "./image"
 export { Tool, ToolFailure, toDefinitions } from "./tool"

+ 43 - 12
packages/ai/src/protocols/google-images.ts

@@ -1,6 +1,13 @@
 import { Effect, Encoding, Schema } from "effect"
 import { Headers, HttpClientRequest } from "effect/unstable/http"
-import { GeneratedImage, ImageModel, ImageResponse, type ImageRequestFor, type ImageRoute } from "../image"
+import {
+  GeneratedImage,
+  ImageModel,
+  ImageResponse,
+  type ImageInput,
+  type ImageRequestFor,
+  type ImageRoute,
+} from "../image"
 import { Auth, type Definition as AuthDefinition } from "../route/auth"
 import {
   InvalidProviderOutputReason,
@@ -12,6 +19,7 @@ import {
   type ProviderMetadata,
 } from "../schema"
 import { ProviderShared } from "./shared"
+import { ImageInputs } from "./utils/image-input"
 
 const ADAPTER = "google-images"
 export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"
@@ -31,7 +39,7 @@ export type GoogleImageOptions = {
 export type GoogleImageBody = Record<string, unknown> & {
   readonly contents: ReadonlyArray<{
     readonly role: "user"
-    readonly parts: ReadonlyArray<{ readonly text: string }>
+    readonly parts: ReadonlyArray<Record<string, unknown>>
   }>
   readonly generationConfig: Record<string, unknown>
 }
@@ -116,15 +124,6 @@ const nativeOptions = (options: GoogleImageOptions | undefined) => {
   )
 }
 
-const body = (request: ImageRequestFor<GoogleImageOptions>, overlay: Record<string, unknown> | undefined) =>
-  mergeJsonRecords(
-    {
-      contents: [{ role: "user", parts: [{ text: request.prompt }] }],
-      generationConfig: nativeOptions(request.options),
-    },
-    overlay,
-  ) as GoogleImageBody
-
 const invalidOutput = (message: string, providerMetadata?: ProviderMetadata) =>
   new LLMError({
     module: ADAPTER,
@@ -143,8 +142,16 @@ export const model = (input: ModelInput) => {
   const route: ImageRoute<GoogleImageOptions> = {
     id: ADAPTER,
     generate: Effect.fn("GoogleImages.generate")(function* (request: ImageRequestFor<GoogleImageOptions>, execute) {
+      const imageParts = yield* Effect.forEach(request.images ?? [], googleImagePart)
       const http = mergeHttpOptions(request.model.http, request.http)
-      const text = ProviderShared.encodeJson(body(request, http?.body))
+      const requestBody = mergeJsonRecords(
+        {
+          contents: [{ role: "user", parts: [{ text: request.prompt }, ...imageParts] }],
+          generationConfig: nativeOptions(request.options),
+        },
+        http?.body,
+      ) as GoogleImageBody
+      const text = ProviderShared.encodeJson(requestBody)
       const url = applyQuery(
         `${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}/models/${request.model.id}:generateContent`,
         http?.query,
@@ -278,6 +285,30 @@ export const model = (input: ModelInput) => {
   return ImageModel.make<GoogleImageOptions>({ id: input.id, provider: "google", route, http: input.http })
 }
 
+const googleImagePart = (image: ImageInput): Effect.Effect<Record<string, unknown>, LLMError> => {
+  if (image.type === "bytes")
+    return Effect.succeed({ inlineData: { mimeType: image.mediaType, data: Encoding.encodeBase64(image.data) } })
+  if (image.type === "file-uri") return Effect.succeed({ fileData: { mimeType: image.mediaType, fileUri: image.uri } })
+  if (image.type === "url")
+    return ImageInputs.decodeDataUrl(image.url, ADAPTER).pipe(
+      Effect.flatMap((decoded) => {
+        if (decoded === undefined)
+          return Effect.fail(
+            ImageInputs.invalid(
+              ADAPTER,
+              "Google generateContent does not fetch public image URLs; use bytes, a data URL, or a Gemini file URI",
+            ),
+          )
+        return Effect.succeed({
+          inlineData: { mimeType: decoded.mediaType, data: Encoding.encodeBase64(decoded.data) },
+        })
+      }),
+    )
+  return Effect.fail(
+    ImageInputs.invalid(ADAPTER, "Google generateContent requires Gemini file URIs rather than provider file IDs"),
+  )
+}
+
 export const GoogleImages = {
   model,
 } as const

+ 151 - 51
packages/ai/src/protocols/openai-images.ts

@@ -1,6 +1,13 @@
 import { Effect, Encoding, Schema } from "effect"
-import { Headers, HttpClientRequest } from "effect/unstable/http"
-import { ImageModel, GeneratedImage, ImageResponse, type ImageRequestFor, type ImageRoute } from "../image"
+import { Headers, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"
+import {
+  ImageModel,
+  GeneratedImage,
+  ImageResponse,
+  type ImageInput,
+  type ImageRequestFor,
+  type ImageRoute,
+} from "../image"
 import { Auth, type Definition as AuthDefinition } from "../route/auth"
 import {
   InvalidProviderOutputReason,
@@ -11,15 +18,18 @@ import {
   type HttpOptions,
 } from "../schema"
 import { ProviderShared } from "./shared"
+import { ImageInputs } from "./utils/image-input"
 import { OpenAIImage } from "./utils/openai-image"
 
 const ADAPTER = "openai-images"
 export const DEFAULT_BASE_URL = "https://api.openai.com/v1"
 export const PATH = "/images/generations"
+export const EDIT_PATH = "/images/edits"
 
 export type OpenAIImageString<Known extends string> = Known | (string & {})
 
 export type OpenAIImageOptions = {
+  readonly mask?: ImageInput
   readonly n?: number
   readonly size?: OpenAIImageString<
     "auto" | "256x256" | "512x512" | "1024x1024" | "1536x1024" | "1024x1536" | "1792x1024" | "1024x1792"
@@ -64,9 +74,9 @@ export interface ModelInput {
   readonly http?: HttpOptions
 }
 
-const nativeOptions = (options: Record<string, unknown> | undefined) => {
+const nativeOptions = (options: OpenAIImageOptions | undefined) => {
   if (!options) return undefined
-  const { outputFormat, outputCompression, ...native } = options
+  const { mask: _, outputFormat, outputCompression, ...native } = options
   return {
     output_format: outputFormat,
     output_compression: outputCompression,
@@ -92,14 +102,89 @@ export const model = (input: ModelInput) => {
   const route: ImageRoute<OpenAIImageOptions> = {
     id: ADAPTER,
     generate: Effect.fn("OpenAIImages.generate")(function* (request: ImageRequestFor<OpenAIImageOptions>, execute) {
+      const mask = request.options?.mask
+      if (mask !== undefined && (request.images?.length ?? 0) === 0)
+        return yield* ImageInputs.invalid(ADAPTER, "An OpenAI image mask requires at least one input image")
       const http = mergeHttpOptions(request.model.http, request.http)
+      const sourceImages = request.images ?? []
+      const multipartImages = yield* Effect.forEach(sourceImages, (image) => {
+        if (image.type === "bytes") return Effect.succeed({ data: image.data, mediaType: image.mediaType })
+        if (image.type === "url") return ImageInputs.decodeDataUrl(image.url, ADAPTER)
+        return Effect.succeed(undefined)
+      })
+      const multipartMask =
+        mask === undefined
+          ? undefined
+          : mask.type === "bytes"
+            ? { data: mask.data, mediaType: mask.mediaType }
+            : mask.type === "url"
+              ? yield* ImageInputs.decodeDataUrl(mask.url, ADAPTER)
+              : undefined
+      const useMultipart =
+        sourceImages.length > 0 &&
+        multipartImages.every((image) => image !== undefined) &&
+        (mask === undefined || multipartMask !== undefined)
+      const path = sourceImages.length === 0 ? PATH : EDIT_PATH
+      const url = applyQuery(`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${path}`, http?.query)
+
+      if (useMultipart) {
+        const form = new FormData()
+        form.append("model", request.model.id)
+        form.append("prompt", request.prompt)
+        Object.entries(mergeJsonRecords(nativeOptions(request.options), http?.body) ?? {}).forEach(([key, value]) => {
+          if (["model", "prompt", "image", "image[]", "images", "mask"].includes(key)) return
+          form.append(key, typeof value === "string" ? value : ProviderShared.encodeJson(value))
+        })
+        multipartImages.forEach((image, index) => {
+          if (image === undefined) return
+          form.append("image[]", imageBlob(image.data, image.mediaType), `image-${index}`)
+        })
+        if (multipartMask !== undefined)
+          form.append("mask", imageBlob(multipartMask.data, multipartMask.mediaType), "mask")
+        const headers = yield* Auth.toEffect(input.auth)({
+          request,
+          method: "POST",
+          url,
+          body: "[multipart/form-data]",
+          headers: Headers.remove(Headers.fromInput({ ...input.headers, ...http?.headers }), "content-type"),
+        })
+        const response = yield* execute(
+          HttpClientRequest.post(url).pipe(HttpClientRequest.setHeaders(headers), HttpClientRequest.bodyFormData(form)),
+        )
+        return yield* parseResponse(response, request.options, http?.body)
+      }
+
+      const references = sourceImages.map((image) => {
+        if (image.type === "bytes") return { image_url: ImageInputs.dataUrl(image) }
+        if (image.type === "url") return { image_url: image.url }
+        if (image.type === "file-id") return { file_id: image.id }
+        return undefined
+      })
+      if (references.some((image) => image === undefined))
+        return yield* ImageInputs.invalid(ADAPTER, "OpenAI Images accepts image URLs, data URLs, bytes, and file IDs")
+      const maskReference =
+        mask === undefined
+          ? undefined
+          : mask.type === "bytes"
+            ? { image_url: ImageInputs.dataUrl(mask) }
+            : mask.type === "url"
+              ? { image_url: mask.url }
+              : mask.type === "file-id"
+                ? { file_id: mask.id }
+                : undefined
+      if (mask !== undefined && maskReference === undefined)
+        return yield* ImageInputs.invalid(ADAPTER, "OpenAI Images accepts masks as URLs, data URLs, bytes, or file IDs")
       const requestBody = mergeJsonRecords(
-        { model: request.model.id, prompt: request.prompt },
+        {
+          model: request.model.id,
+          prompt: request.prompt,
+          images: references.length === 0 ? undefined : references,
+          mask: maskReference,
+        },
         nativeOptions(request.options),
         http?.body,
       ) as OpenAIImageBody
       const text = ProviderShared.encodeJson(requestBody)
-      const url = applyQuery(`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${PATH}`, http?.query)
       const headers = yield* Auth.toEffect(input.auth)({
         request,
         method: "POST",
@@ -113,56 +198,71 @@ export const model = (input: ModelInput) => {
           HttpClientRequest.bodyText(text, "application/json"),
         ),
       )
-      const payload = yield* response.json.pipe(
-        Effect.mapError(() => invalidOutput("Failed to read the OpenAI Images response")),
-      )
-      const decoded = yield* Schema.decodeUnknownEffect(OpenAIImageResponse)(payload).pipe(
-        Effect.mapError(() => invalidOutput("OpenAI Images returned an invalid response")),
-      )
-      const format =
-        decoded.output_format ?? (typeof requestBody.output_format === "string" ? requestBody.output_format : "png")
-      const images = yield* Effect.forEach(decoded.data, (item, index) => {
-        if (item.b64_json)
-          return Effect.fromResult(Encoding.decodeBase64(item.b64_json)).pipe(
-            Effect.mapError(() => invalidOutput(`OpenAI Images result ${index} contains invalid base64 data`)),
-            Effect.map(
-              (data) =>
-                new GeneratedImage({
-                  mediaType: `image/${format}`,
-                  data,
-                  providerMetadata:
-                    item.revised_prompt === undefined ? undefined : { openai: { revisedPrompt: item.revised_prompt } },
-                }),
-            ),
-          )
-        if (item.url)
-          return Effect.succeed(
+      return yield* parseResponse(response, request.options, http?.body)
+    }),
+  }
+  return ImageModel.make<OpenAIImageOptions>({ id: input.id, provider: "openai", route, http: input.http })
+}
+
+const parseResponse = Effect.fn("OpenAIImages.parseResponse")(function* (
+  response: HttpClientResponse.HttpClientResponse,
+  options: OpenAIImageOptions | undefined,
+  overlay: Record<string, unknown> | undefined,
+) {
+  const payload = yield* response.json.pipe(
+    Effect.mapError(() => invalidOutput("Failed to read the OpenAI Images response")),
+  )
+  const decoded = yield* Schema.decodeUnknownEffect(OpenAIImageResponse)(payload).pipe(
+    Effect.mapError(() => invalidOutput("OpenAI Images returned an invalid response")),
+  )
+  const requestBody = mergeJsonRecords(nativeOptions(options), overlay)
+  const format =
+    decoded.output_format ?? (typeof requestBody?.output_format === "string" ? requestBody.output_format : "png")
+  const images = yield* Effect.forEach(decoded.data, (item, index) => {
+    if (item.b64_json)
+      return Effect.fromResult(Encoding.decodeBase64(item.b64_json)).pipe(
+        Effect.mapError(() => invalidOutput(`OpenAI Images result ${index} contains invalid base64 data`)),
+        Effect.map(
+          (data) =>
             new GeneratedImage({
               mediaType: `image/${format}`,
-              data: item.url,
+              data,
               providerMetadata:
                 item.revised_prompt === undefined ? undefined : { openai: { revisedPrompt: item.revised_prompt } },
             }),
-          )
-        return Effect.fail(invalidOutput(`OpenAI Images result ${index} has neither image data nor a URL`))
-      })
-      if (images.length === 0) return yield* invalidOutput("OpenAI Images returned no images")
-      return new ImageResponse({
-        images,
-        usage:
-          decoded.usage === undefined
-            ? undefined
-            : new Usage({
-                inputTokens: decoded.usage.input_tokens,
-                outputTokens: decoded.usage.output_tokens,
-                totalTokens: decoded.usage.total_tokens,
-                providerMetadata: { openai: decoded.usage },
-              }),
-        providerMetadata: { openai: { outputFormat: format } },
-      })
-    }),
-  }
-  return ImageModel.make<OpenAIImageOptions>({ id: input.id, provider: "openai", route, http: input.http })
+        ),
+      )
+    if (item.url)
+      return Effect.succeed(
+        new GeneratedImage({
+          mediaType: `image/${format}`,
+          data: item.url,
+          providerMetadata:
+            item.revised_prompt === undefined ? undefined : { openai: { revisedPrompt: item.revised_prompt } },
+        }),
+      )
+    return Effect.fail(invalidOutput(`OpenAI Images result ${index} has neither image data nor a URL`))
+  })
+  if (images.length === 0) return yield* invalidOutput("OpenAI Images returned no images")
+  return new ImageResponse({
+    images,
+    usage:
+      decoded.usage === undefined
+        ? undefined
+        : new Usage({
+            inputTokens: decoded.usage.input_tokens,
+            outputTokens: decoded.usage.output_tokens,
+            totalTokens: decoded.usage.total_tokens,
+            providerMetadata: { openai: decoded.usage },
+          }),
+    providerMetadata: { openai: { outputFormat: format } },
+  })
+})
+
+const imageBlob = (data: Uint8Array, mediaType: string) => {
+  const buffer = new ArrayBuffer(data.byteLength)
+  new Uint8Array(buffer).set(data)
+  return new Blob([buffer], { type: mediaType })
 }
 
 export const OpenAIImages = {

+ 34 - 0
packages/ai/src/protocols/utils/image-input.ts

@@ -0,0 +1,34 @@
+import { Effect, Encoding } from "effect"
+import type { ImageInput } from "../../image"
+import { InvalidRequestReason, LLMError } from "../../schema"
+
+const invalid = (module: string, message: string) =>
+  new LLMError({
+    module,
+    method: "generate",
+    reason: new InvalidRequestReason({ message }),
+  })
+
+export const dataUrl = (input: Extract<ImageInput, { readonly type: "bytes" }>) =>
+  `data:${input.mediaType};base64,${Encoding.encodeBase64(input.data)}`
+
+export const decodeDataUrl = (
+  url: string,
+  module: string,
+): Effect.Effect<{ readonly mediaType: string; readonly data: Uint8Array } | undefined, LLMError> => {
+  if (!url.startsWith("data:")) return Effect.succeed(undefined)
+  const match = /^data:([^;,]+);base64,(.*)$/s.exec(url)
+  if (!match) return Effect.fail(invalid(module, "Image data URLs must contain a MIME type and base64 data"))
+  return Effect.fromResult(Encoding.decodeBase64(match[2])).pipe(
+    Effect.mapError(() => invalid(module, "Image data URL contains invalid base64 data")),
+    Effect.map((data) => ({ mediaType: match[1], data })),
+  )
+}
+
+export const invalidImageInput = invalid
+
+export const ImageInputs = {
+  dataUrl,
+  decodeDataUrl,
+  invalid: invalidImageInput,
+} as const

+ 20 - 2
packages/ai/src/protocols/xai-images.ts

@@ -11,10 +11,12 @@ import {
   type HttpOptions,
 } from "../schema"
 import { ProviderShared, optionalNull } from "./shared"
+import { ImageInputs } from "./utils/image-input"
 
 const ADAPTER = "xai-images"
 export const DEFAULT_BASE_URL = "https://api.x.ai/v1"
 export const PATH = "/images/generations"
+export const EDIT_PATH = "/images/edits"
 
 export type XAIImageString<Known extends string> = Known | (string & {})
 
@@ -111,13 +113,29 @@ export const model = (input: ModelInput) => {
     id: ADAPTER,
     generate: Effect.fn("XAIImages.generate")(function* (request: ImageRequestFor<XAIImageOptions>, execute) {
       const http = mergeHttpOptions(request.model.http, request.http)
+      const imageReferences = (request.images ?? []).map((image) => {
+        if (image.type === "bytes") return { url: ImageInputs.dataUrl(image), type: "image_url" as const }
+        if (image.type === "url") return { url: image.url, type: "image_url" as const }
+        if (image.type === "file-id") return { file_id: image.id }
+        return undefined
+      })
+      if (imageReferences.some((image) => image === undefined))
+        return yield* ImageInputs.invalid(ADAPTER, "xAI Images accepts image URLs, data URLs, bytes, and file IDs")
       const requestBody = mergeJsonRecords(
-        { model: request.model.id, prompt: request.prompt },
+        {
+          model: request.model.id,
+          prompt: request.prompt,
+          image: imageReferences.length === 1 ? imageReferences[0] : undefined,
+          images: imageReferences.length > 1 ? imageReferences : undefined,
+        },
         nativeOptions(request.options),
         http?.body,
       ) as XAIImageBody
       const text = ProviderShared.encodeJson(requestBody)
-      const url = applyQuery(`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${PATH}`, http?.query)
+      const url = applyQuery(
+        `${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${imageReferences.length === 0 ? PATH : EDIT_PATH}`,
+        http?.query,
+      )
       const headers = yield* Auth.toEffect(input.auth)({
         request,
         method: "POST",

+ 3 - 0
packages/ai/src/protocols/zai-images.ts

@@ -4,6 +4,7 @@ import { GeneratedImage, ImageModel, ImageResponse, type ImageRequestFor, type I
 import { Auth, type Definition as AuthDefinition } from "../route/auth"
 import { InvalidProviderOutputReason, LLMError, mergeHttpOptions, mergeJsonRecords, type HttpOptions } from "../schema"
 import { ProviderShared } from "./shared"
+import { ImageInputs } from "./utils/image-input"
 
 const ADAPTER = "zai-images"
 export const DEFAULT_BASE_URL = "https://api.z.ai/api/paas/v4"
@@ -74,6 +75,8 @@ export const model = (input: ModelInput) => {
   const route: ImageRoute<ZAIImageOptions> = {
     id: ADAPTER,
     generate: Effect.fn("ZAIImages.generate")(function* (request: ImageRequestFor<ZAIImageOptions>, execute) {
+      if ((request.images?.length ?? 0) > 0)
+        return yield* ImageInputs.invalid(ADAPTER, "Z.ai hosted image generation does not support image inputs")
       const http = mergeHttpOptions(request.model.http, request.http)
       const requestBody = mergeJsonRecords(
         { model: request.model.id, prompt: request.prompt },

+ 3 - 7
packages/ai/test/exports.test.ts

@@ -1,5 +1,5 @@
 import { describe, expect, test } from "bun:test"
-import { LLM, LLMClient, Provider } from "@opencode-ai/ai"
+import { ImageInput, LLM, LLMClient, Provider } from "@opencode-ai/ai"
 import { Route, Protocol } from "@opencode-ai/ai/route"
 import { Provider as ProviderSubpath } from "@opencode-ai/ai/provider"
 import {
@@ -11,12 +11,7 @@ import {
   XAI,
 } from "@opencode-ai/ai/providers"
 import * as GitHubCopilot from "@opencode-ai/ai/providers/github-copilot"
-import {
-  OpenAIChat,
-  OpenAICompatibleChat,
-  OpenAICompatibleResponses,
-  OpenAIResponses,
-} from "@opencode-ai/ai/protocols"
+import { OpenAIChat, OpenAICompatibleChat, OpenAICompatibleResponses, OpenAIResponses } from "@opencode-ai/ai/protocols"
 import * as AnthropicMessages from "@opencode-ai/ai/protocols/anthropic-messages"
 
 describe("public exports", () => {
@@ -24,6 +19,7 @@ describe("public exports", () => {
     expect(LLM.request).toBeFunction()
     expect(LLMClient.Service).toBeFunction()
     expect(LLMClient.layer).toBeDefined()
+    expect(ImageInput.bytes).toBeFunction()
     expect(Provider.make).toBeFunction()
     expect(ProviderSubpath.make).toBe(Provider.make)
   })

BIN
packages/ai/test/fixtures/images/edit-source.jpg


File diff suppressed because it is too large
+ 23 - 0
packages/ai/test/fixtures/recordings/google-images/edits-an-image.json


File diff suppressed because it is too large
+ 16 - 0
packages/ai/test/fixtures/recordings/openai-images/edits-an-image.json


File diff suppressed because it is too large
+ 23 - 0
packages/ai/test/fixtures/recordings/xai-images/edits-an-image.json


+ 207 - 2
packages/ai/test/image.test.ts

@@ -1,8 +1,8 @@
 import { describe, expect } from "bun:test"
 import { Effect, Layer } from "effect"
 import { HttpClientRequest } from "effect/unstable/http"
-import { Image, ImageClient } from "../src"
-import { Google, OpenAI } from "../src/providers"
+import { Image, ImageClient, ImageInput } from "../src"
+import { Google, OpenAI, XAI, ZAI } from "../src/providers"
 import { it } from "./lib/effect"
 import { dynamicResponse } from "./lib/http"
 
@@ -125,6 +125,211 @@ describe("Image", () => {
     ),
   )
 
+  it.effect("routes OpenAI byte inputs and masks through multipart edits", () =>
+    Image.generate({
+      model: OpenAI.configure({ apiKey: "test", baseURL: "https://api.openai.test/v1" }).image("future-model"),
+      prompt: "Combine these images",
+      images: [
+        ImageInput.bytes(Uint8Array.from([1, 2, 3]), "image/png"),
+        ImageInput.url("data:image/jpeg;base64,BAUG"),
+      ],
+      options: {
+        mask: ImageInput.bytes(Uint8Array.from([7, 8, 9]), "image/png"),
+        quality: "high",
+        future_option: true,
+      },
+      http: {
+        body: { quality: "low", model: "corrupt", prompt: "corrupt", image: "corrupt", "image[]": "corrupt" },
+        headers: { "content-type": "application/json" },
+      },
+    }).pipe(
+      Effect.provide(
+        ImageClient.layer.pipe(
+          Layer.provide(
+            dynamicResponse((input) =>
+              Effect.gen(function* () {
+                const request = yield* HttpClientRequest.toWeb(input.request).pipe(Effect.orDie)
+                expect(request.url).toBe("https://api.openai.test/v1/images/edits")
+                expect(request.headers.get("content-type")).toStartWith("multipart/form-data; boundary=")
+                expect(input.text).toContain('name="model"\r\n\r\nfuture-model')
+                expect(input.text).toContain('name="prompt"\r\n\r\nCombine these images')
+                expect(input.text.match(/name="image\[\]"/g)).toHaveLength(2)
+                expect(input.text).toContain('name="mask"')
+                expect(input.text).toContain('name="quality"\r\n\r\nlow')
+                expect(input.text).not.toContain("corrupt")
+                return input.respond(JSON.stringify({ data: [{ b64_json: "AQID" }] }), {
+                  headers: { "content-type": "application/json" },
+                })
+              }),
+            ),
+          ),
+        ),
+      ),
+    ),
+  )
+
+  it.effect("routes OpenAI URL and file inputs through JSON edits", () =>
+    Image.generate({
+      model: OpenAI.configure({ apiKey: "test", baseURL: "https://api.openai.test/v1" }).image("future-model"),
+      prompt: "Combine these images",
+      images: [ImageInput.url("https://example.test/source.png"), ImageInput.file("file_123")],
+      options: { mask: ImageInput.file("file_mask") },
+      http: { body: { future_option: true } },
+    }).pipe(
+      Effect.provide(
+        ImageClient.layer.pipe(
+          Layer.provide(
+            dynamicResponse((input) => {
+              expect(JSON.parse(input.text)).toEqual({
+                model: "future-model",
+                prompt: "Combine these images",
+                images: [{ image_url: "https://example.test/source.png" }, { file_id: "file_123" }],
+                mask: { file_id: "file_mask" },
+                future_option: true,
+              })
+              return Effect.succeed(
+                input.respond(JSON.stringify({ data: [{ b64_json: "AQID" }] }), {
+                  headers: { "content-type": "application/json" },
+                }),
+              )
+            }),
+          ),
+        ),
+      ),
+    ),
+  )
+
+  it.effect("routes ordered xAI image inputs through JSON edits", () =>
+    Image.generate({
+      model: XAI.configure({ apiKey: "test", baseURL: "https://api.xai.test/v1" }).image("future-model"),
+      prompt: "Combine these images",
+      images: [
+        ImageInput.bytes(Uint8Array.from([1, 2, 3]), "image/png"),
+        ImageInput.url("https://example.test/source.jpg"),
+        ImageInput.file("file_123"),
+      ],
+    }).pipe(
+      Effect.provide(
+        ImageClient.layer.pipe(
+          Layer.provide(
+            dynamicResponse((input) => {
+              expect(JSON.parse(input.text)).toEqual({
+                model: "future-model",
+                prompt: "Combine these images",
+                images: [
+                  { url: "data:image/png;base64,AQID", type: "image_url" },
+                  { url: "https://example.test/source.jpg", type: "image_url" },
+                  { file_id: "file_123" },
+                ],
+              })
+              return Effect.succeed(
+                input.respond(JSON.stringify({ data: [{ b64_json: "AQID", mime_type: "image/png" }] }), {
+                  headers: { "content-type": "application/json" },
+                }),
+              )
+            }),
+          ),
+        ),
+      ),
+    ),
+  )
+
+  it.effect("uses xAI's singular image field for one input", () =>
+    Image.generate({
+      model: XAI.configure({ apiKey: "test", baseURL: "https://api.xai.test/v1" }).image("future-model"),
+      prompt: "Edit this image",
+      images: [ImageInput.file("file_123")],
+    }).pipe(
+      Effect.provide(
+        ImageClient.layer.pipe(
+          Layer.provide(
+            dynamicResponse((input) => {
+              expect(JSON.parse(input.text)).toEqual({
+                model: "future-model",
+                prompt: "Edit this image",
+                image: { file_id: "file_123" },
+              })
+              return Effect.succeed(
+                input.respond(JSON.stringify({ data: [{ b64_json: "AQID", mime_type: "image/png" }] }), {
+                  headers: { "content-type": "application/json" },
+                }),
+              )
+            }),
+          ),
+        ),
+      ),
+    ),
+  )
+
+  it.effect("lowers ordered Google image inputs into generateContent parts", () =>
+    Image.generate({
+      model: Google.configure({ apiKey: "test", baseURL: "https://google.test/v1beta" }).image("future-model"),
+      prompt: "Combine these images",
+      images: [
+        ImageInput.bytes(Uint8Array.from([1, 2, 3]), "image/png"),
+        ImageInput.url("data:image/jpeg;base64,BAUG"),
+        ImageInput.fileUri("https://generativelanguage.googleapis.com/v1beta/files/123", "image/webp"),
+      ],
+    }).pipe(
+      Effect.provide(
+        ImageClient.layer.pipe(
+          Layer.provide(
+            dynamicResponse((input) => {
+              expect(JSON.parse(input.text).contents[0].parts).toEqual([
+                { text: "Combine these images" },
+                { inlineData: { mimeType: "image/png", data: "AQID" } },
+                { inlineData: { mimeType: "image/jpeg", data: "BAUG" } },
+                {
+                  fileData: {
+                    mimeType: "image/webp",
+                    fileUri: "https://generativelanguage.googleapis.com/v1beta/files/123",
+                  },
+                },
+              ])
+              return Effect.succeed(
+                input.respond(
+                  JSON.stringify({
+                    candidates: [{ content: { parts: [{ inlineData: { mimeType: "image/png", data: "AQID" } }] } }],
+                  }),
+                  { headers: { "content-type": "application/json" } },
+                ),
+              )
+            }),
+          ),
+        ),
+      ),
+    ),
+  )
+
+  it.effect("rejects unsupported provider inputs before sending", () =>
+    Effect.gen(function* () {
+      const cases = [
+        Image.generate({
+          model: Google.configure({ apiKey: "test" }).image("model"),
+          prompt: "edit",
+          images: [ImageInput.url("https://example.test/image.png")],
+        }),
+        Image.generate({
+          model: ZAI.configure({ apiKey: "test" }).image("model"),
+          prompt: "edit",
+          images: [ImageInput.bytes(Uint8Array.from([1]), "image/png")],
+        }),
+      ]
+      yield* Effect.forEach(cases, (program) =>
+        program.pipe(
+          Effect.flip,
+          Effect.tap((error) => Effect.sync(() => expect(error.reason._tag).toBe("InvalidRequest"))),
+        ),
+      )
+    }).pipe(
+      Effect.provide(
+        ImageClient.layer.pipe(
+          Layer.provide(dynamicResponse(() => Effect.die("unsupported input reached the network"))),
+        ),
+      ),
+    ),
+  )
+
   it.effect("generates images through the Google generateContent API", () =>
     Effect.gen(function* () {
       const response = yield* Image.generate({

+ 26 - 1
packages/ai/test/image.types.ts

@@ -1,5 +1,6 @@
 import {
   Image,
+  ImageInput,
   ImageModel,
   type ImageModelOptions,
   type ImageOptions,
@@ -22,6 +23,11 @@ void invalidGoogleOptions
 Image.generate({
   model: google,
   prompt: "A lighthouse",
+  images: [
+    ImageInput.bytes(Uint8Array.from([1, 2, 3]), "image/png"),
+    ImageInput.url("data:image/jpeg;base64,AQID"),
+    ImageInput.fileUri("https://generativelanguage.googleapis.com/v1beta/files/example", "image/webp"),
+  ],
   options: { aspectRatio: "16:9", imageSize: "2K", futureOption: true },
 })
 
@@ -60,7 +66,14 @@ void futureOpenAIOptions
 Image.generate({
   model: openai,
   prompt: "A lighthouse",
-  options: { quality: "hd", outputFormat: "webp", size: "2048x2048", future_option: true },
+  images: [ImageInput.url("https://example.com/source.png"), ImageInput.file("file_123")],
+  options: {
+    mask: ImageInput.bytes(Uint8Array.from([1]), "image/png"),
+    quality: "hd",
+    outputFormat: "webp",
+    size: "2048x2048",
+    future_option: true,
+  },
 })
 Image.generate({ model: openai, prompt: "A lighthouse", options: { quality: "future-quality", size: "256x256" } })
 Image.generate({ model: openai, prompt: "A lighthouse", options: { size: "1792x1024" } })
@@ -81,6 +94,7 @@ XAI.configure({ image: { options: { resolution: "1k" } } })
 Image.generate({
   model: xai,
   prompt: "A lighthouse",
+  images: [ImageInput.url("data:image/png;base64,AQID"), ImageInput.file("file_123")],
   options: {
     n: 2,
     aspectRatio: "future-ratio",
@@ -115,6 +129,15 @@ Image.generate({ model: zai, prompt: "A lighthouse", options: { userID: 1 } })
 
 declare const generic: ImageModel<ImageOptions>
 Image.generate({ model: generic, prompt: "A lighthouse", options: { arbitrary: true } })
+const explicitImageInput: ImageInput = ImageInput.url("https://example.com/image.png")
+void explicitImageInput
+
+// @ts-expect-error Raw strings are ambiguous and are not image inputs.
+Image.generate({ model: openai, prompt: "A lighthouse", images: ["AQID"] })
+// @ts-expect-error Byte image inputs require an explicit MIME type.
+Image.generate({ model: openai, prompt: "A lighthouse", images: [{ type: "bytes", data: new Uint8Array() }] })
+// @ts-expect-error File URIs require an explicit MIME type for Gemini fileData.
+Image.generate({ model: google, prompt: "A lighthouse", images: [{ type: "file-uri", uri: "files/123" }] })
 
 const request = Image.request({
   model: google,
@@ -134,3 +157,5 @@ Image.generate({ model: openai, prompt: "A lighthouse", aspectRatio: "16:9" })
 Image.generate({ model: openai, prompt: "A lighthouse", seed: 1 })
 // @ts-expect-error Image requests do not expose metadata.
 Image.generate({ model: openai, prompt: "A lighthouse", metadata: { trace: true } })
+// @ts-expect-error Masks are provider options, not a common image request field.
+Image.generate({ model: openai, prompt: "A lighthouse", mask: ImageInput.url("https://example.com/mask.png") })

+ 29 - 0
packages/ai/test/lib/image.ts

@@ -0,0 +1,29 @@
+export const dimensions = (data: Uint8Array) => {
+  if (data[0] === 0x89 && data[1] === 0x50 && data[2] === 0x4e && data[3] === 0x47)
+    return {
+      width: readUint32(data, 16),
+      height: readUint32(data, 20),
+    }
+  if (data[0] === 0xff && data[1] === 0xd8) {
+    for (let offset = 2; offset + 8 < data.length; ) {
+      if (data[offset] !== 0xff) {
+        offset++
+        continue
+      }
+      const marker = data[offset + 1]
+      if (
+        marker !== undefined &&
+        [0xc0, 0xc1, 0xc2, 0xc3, 0xc5, 0xc6, 0xc7, 0xc9, 0xca, 0xcb, 0xcd, 0xce, 0xcf].includes(marker)
+      )
+        return {
+          width: (data[offset + 7] << 8) | data[offset + 8],
+          height: (data[offset + 5] << 8) | data[offset + 6],
+        }
+      offset += 2 + ((data[offset + 2] << 8) | data[offset + 3])
+    }
+  }
+  throw new Error("Unsupported image fixture format")
+}
+
+const readUint32 = (data: Uint8Array, offset: number) =>
+  ((data[offset] << 24) | (data[offset + 1] << 16) | (data[offset + 2] << 8) | data[offset + 3]) >>> 0

+ 24 - 1
packages/ai/test/provider/google-images.recorded.test.ts

@@ -1,7 +1,8 @@
 import { describe, expect } from "bun:test"
 import { Effect } from "effect"
-import { Image } from "../../src"
+import { Image, ImageInput } from "../../src"
 import { Google } from "../../src/providers"
+import { dimensions } from "../lib/image"
 import { recordedTests } from "../recorded-test"
 
 const model = Google.configure({
@@ -30,4 +31,26 @@ describe("Google Images recorded", () => {
       expect(response.image?.data.length).toBeGreaterThan(0)
     }),
   )
+
+  recorded.effect("edits an image", () =>
+    Effect.gen(function* () {
+      const response = yield* Image.generate({
+        model,
+        prompt:
+          "Transform this minimal source into a bright orange sun icon with eight rounded rays on a pale blue background.",
+        images: [
+          ImageInput.bytes(
+            yield* Effect.promise(() => Bun.file("test/fixtures/images/edit-source.jpg").bytes()),
+            "image/jpeg",
+          ),
+        ],
+        options: { aspectRatio: "1:1" },
+      })
+
+      expect(response.image?.mediaType).toBe("image/jpeg")
+      expect(response.image?.data).toBeInstanceOf(Uint8Array)
+      if (!(response.image?.data instanceof Uint8Array)) throw new Error("Expected owned Google image bytes")
+      expect(dimensions(response.image.data)).toEqual({ width: 1024, height: 1024 })
+    }),
+  )
 })

+ 30 - 1
packages/ai/test/provider/openai-images.recorded.test.ts

@@ -1,7 +1,8 @@
 import { describe, expect } from "bun:test"
 import { Effect } from "effect"
-import { Image } from "../../src"
+import { Image, ImageInput } from "../../src"
 import { OpenAI } from "../../src/providers"
+import { dimensions } from "../lib/image"
 import { recordedTests } from "../recorded-test"
 
 const model = OpenAI.configure({
@@ -30,4 +31,32 @@ describe("OpenAI Images recorded", () => {
       expect(response.image?.data.length).toBeGreaterThan(0)
     }),
   )
+
+  recorded.effect.with(
+    "edits an image",
+    {
+      options: {
+        match: (incoming, recorded) => incoming.method === recorded.method && incoming.url === recorded.url,
+      },
+    },
+    () =>
+      Effect.gen(function* () {
+        const response = yield* Image.generate({
+          model,
+          prompt: "Keep the simple shape and change it from black to bright green.",
+          images: [
+            ImageInput.bytes(
+              yield* Effect.promise(() => Bun.file("test/fixtures/images/edit-source.jpg").bytes()),
+              "image/jpeg",
+            ),
+          ],
+          options: { quality: "low", outputFormat: "jpeg", outputCompression: 10, size: "1024x1024" },
+        })
+
+        expect(response.image?.mediaType).toBe("image/jpeg")
+        expect(response.image?.data).toBeInstanceOf(Uint8Array)
+        if (!(response.image?.data instanceof Uint8Array)) throw new Error("Expected owned OpenAI image bytes")
+        expect(dimensions(response.image.data)).toEqual({ width: 1024, height: 1024 })
+      }),
+  )
 })

+ 23 - 1
packages/ai/test/provider/xai-images.recorded.test.ts

@@ -1,7 +1,8 @@
 import { describe, expect } from "bun:test"
 import { Effect } from "effect"
-import { Image } from "../../src"
+import { Image, ImageInput } from "../../src"
 import { XAI } from "../../src/providers"
+import { dimensions } from "../lib/image"
 import { recordedTests } from "../recorded-test"
 
 const model = XAI.configure({
@@ -30,4 +31,25 @@ describe("xAI Images recorded", () => {
       expect(response.image?.data.length).toBeGreaterThan(0)
     }),
   )
+
+  recorded.effect("edits an image", () =>
+    Effect.gen(function* () {
+      const response = yield* Image.generate({
+        model,
+        prompt: "Keep the simple shape and change it from black to bright purple.",
+        images: [
+          ImageInput.bytes(
+            yield* Effect.promise(() => Bun.file("test/fixtures/images/edit-source.jpg").bytes()),
+            "image/jpeg",
+          ),
+        ],
+        options: { aspectRatio: "1:1", resolution: "1k", responseFormat: "b64_json" },
+      })
+
+      expect(response.image?.mediaType).toMatch(/^image\/(jpeg|png)$/)
+      expect(response.image?.data).toBeInstanceOf(Uint8Array)
+      if (!(response.image?.data instanceof Uint8Array)) throw new Error("Expected owned xAI image bytes")
+      expect(dimensions(response.image.data)).toEqual({ width: 1024, height: 1024 })
+    }),
+  )
 })

Some files were not shown because too many files changed in this diff