fix(imagegen): separate generation and editing tools

This commit is contained in:
程序员阿江(Relakkes)
2026-08-05 03:30:50 +08:00
parent 8755a03f57
commit ba5dcffb22
14 changed files with 281 additions and 75 deletions
@@ -48,6 +48,9 @@ export function ImageGenerationBlock({
const slotCount = requestedCount ?? parsedResult?.images.length ?? 1
const prompt = stringValue(inputRecord.prompt) ?? parsedResult?.prompt ?? ''
const isEdit = (
Array.isArray(inputRecord.referenced_image_paths) && inputRecord.referenced_image_paths.length > 0
) || (
// Persisted calls from before the official imagegen argument alignment.
Array.isArray(inputRecord.input_images) && inputRecord.input_images.length > 0
) || parsedResult?.operation === 'edit'
const isWaiting = !result
@@ -8,6 +8,7 @@ import { useTranslation } from '../../i18n'
import type { TranslationKey } from '../../i18n'
import { InlineImageGallery } from './InlineImageGallery'
import { ImageGenerationBlock } from './ImageGenerationBlock'
import { isImageGenerationToolName } from './imageGenerationTools'
import type { AgentTaskNotification } from '../../types/chat'
import {
PlanPreviewCard,
@@ -187,7 +188,7 @@ export const ToolCallBlock = memo(function ToolCallBlock({ toolName, input, resu
)
}
if (toolName === 'ImageGen') {
if (isImageGenerationToolName(toolName)) {
return (
<ImageGenerationBlock
input={input}
@@ -1,6 +1,7 @@
import { memo, useCallback, useState } from 'react'
import { BookMarked, ChevronDown, ChevronRight, CircleCheck, Settings } from 'lucide-react'
import { ToolCallBlock } from './ToolCallBlock'
import { isImageGenerationToolName } from './imageGenerationTools'
import { MarkdownRenderer } from '../markdown/MarkdownRenderer'
import { Badge, StatusDot, type Tone } from '@/components/ui/Badge'
import { Button } from '@/components/ui/Button'
@@ -203,8 +204,8 @@ function ToolCallGroupContent({
showOpenRun = true,
isStreaming,
}: Props) {
const hasImageGeneration = toolCalls.some((toolCall) => toolCall.toolName === 'ImageGen')
if (hasImageGeneration && !toolCalls.every((toolCall) => toolCall.toolName === 'ImageGen')) {
const hasImageGeneration = toolCalls.some((toolCall) => isImageGenerationToolName(toolCall.toolName))
if (hasImageGeneration && !toolCalls.every((toolCall) => isImageGenerationToolName(toolCall.toolName))) {
const segments: Array<
| { kind: 'image'; toolCall: ToolCall }
| { kind: 'regular'; toolCalls: ToolCall[] }
@@ -217,7 +218,7 @@ function ToolCallGroupContent({
}
for (const toolCall of toolCalls) {
if (toolCall.toolName === 'ImageGen') {
if (isImageGenerationToolName(toolCall.toolName)) {
flushRegularCalls()
segments.push({ kind: 'image', toolCall })
} else {
@@ -268,7 +269,7 @@ function ToolCallGroupContent({
)
}
const allImageGeneration = toolCalls.length > 0 && toolCalls.every((toolCall) => toolCall.toolName === 'ImageGen')
const allImageGeneration = toolCalls.length > 0 && toolCalls.every((toolCall) => isImageGenerationToolName(toolCall.toolName))
if (allImageGeneration) {
return (
<div className="space-y-2">
@@ -47,14 +47,14 @@ describe('chat blocks', () => {
expect(screen.getAllByText('Generating 2 images')).toHaveLength(2)
})
it('labels an input-image turn as editing while keeping output placeholders', () => {
it('labels a referenced-image turn as editing while keeping output placeholders', () => {
render(
<ToolCallBlock
toolName="ImageGen"
toolName="ImageEdit"
input={{
prompt: 'Change only the scarf color',
count: 2,
input_images: [{ path: '/staged/fox.png', role: 'edit_target' }],
referenced_image_paths: ['/staged/fox.png'],
}}
isPending
/>,
@@ -99,6 +99,45 @@ describe('chat blocks', () => {
expect(screen.queryByRole('button', { name: /ToolSearch \(1\), ImageGen \(1\)/ })).toBeNull()
})
it('keeps image editing outside a mixed deferred-tool summary', () => {
const toolCalls: Array<Extract<UIMessage, { type: 'tool_use' }>> = [
{
id: 'search-use',
type: 'tool_use',
toolName: 'ToolSearch',
toolUseId: 'search-1',
input: { query: 'image editing' },
timestamp: 1,
},
{
id: 'edit-use',
type: 'tool_use',
toolName: 'ImageEdit',
toolUseId: 'edit-1',
input: {
prompt: 'Change only the scarf color',
count: 1,
referenced_image_paths: ['/staged/fox.png'],
},
timestamp: 2,
isPending: true,
},
]
render(
<ToolCallGroup
toolCalls={toolCalls}
resultMap={new Map()}
childToolCallsByParent={new Map()}
agentTaskNotifications={{}}
isStreaming
/>,
)
expect(screen.getAllByTestId('image-generation-slot')).toHaveLength(1)
expect(screen.queryByRole('button', { name: /ToolSearch \(1\), ImageEdit \(1\)/ })).toBeNull()
})
it('replaces every image placeholder with the saved tool result', () => {
const content = JSON.stringify({
type: 'image_generation_result',
@@ -0,0 +1,3 @@
export function isImageGenerationToolName(toolName: string): boolean {
return toolName === 'ImageGen' || toolName === 'ImageEdit'
}
+7 -4
View File
@@ -39,7 +39,7 @@ describe('bundled imagegen skill', () => {
const skill = getBundledSkills().find((command) => command.name === 'imagegen')
expect(skill).toBeDefined()
expect(skill?.allowedTools).toEqual(['ImageGen'])
expect(skill?.allowedTools).toEqual(['ImageGen', 'ImageEdit'])
expect(skill?.isEnabled?.()).toBe(true)
if (!skill || skill.type !== 'prompt') return
@@ -53,12 +53,15 @@ describe('bundled imagegen skill', () => {
.join('\n')
expect(text).toContain('Create two poster variants')
expect(text).toContain('ImageGen')
expect(text).toContain('do not retry <code>ImageGen</code> automatically')
expect(text).toContain('ImageEdit')
expect(text).toContain('do not retry the image tool automatically')
expect(text).toContain('do not repeat, link, or embed the returned local paths')
expect(text).toContain('latest selected output as the next turn')
expect(text).toContain('one call per image')
expect(text).toContain('omit <code>input_images</code> entirely')
expect(text).toContain('Never invent a path')
expect(text).toContain('schema intentionally has no image-path argument')
expect(text).toContain('<code>ImageEdit</code> requires <code>referenced_image_paths</code>')
expect(text).toContain('Never invent, search for, or substitute another filesystem path')
expect(text).toContain('Preserve all relevant user-specified detail')
expect(text).toContain('Provider and image model selection come from')
expect(text).toContain('do not add either to the tool arguments')
expect(text).not.toContain('CC_HAHA_IMAGE_API_KEY')
+1 -1
View File
@@ -14,7 +14,7 @@ export function registerImagegenSkill(): void {
registerBundledSkill({
name: 'imagegen',
description: DESCRIPTION,
allowedTools: ['ImageGen'],
allowedTools: ['ImageGen', 'ImageEdit'],
userInvocable: true,
isEnabled: () => getImageGenerationRuntimeConfig() !== null,
async getPromptForCommand(args) {
+7 -8
View File
@@ -1,30 +1,29 @@
---
name: imagegen
description: Generate original images, artwork, product visuals, diagrams, or other raster assets with the desktop's configured image provider. Use whenever the user asks to create or generate an image.
allowed-tools: ImageGen
allowed-tools: ImageGen, ImageEdit
---
# Image generation
Use the built-in `ImageGen` tool. Provider authentication, model routing, output storage, and secrets are managed by the desktop host; never ask the user to put an API key in this skill or in the prompt.
Use the built-in `ImageGen` and `ImageEdit` tools. Provider authentication, model routing, output storage, and secrets are managed by the desktop host; never ask the user to put an API key in this skill or in the prompt.
## Decide the request shape
- Treat a brand-new visual as generation. Treat a request that preserves or changes an existing visual as an edit.
- For generation, omit `input_images` entirely. Never invent a path to represent the absence of an input image.
- Treat a brand-new visual as generation and call `ImageGen`. Its schema intentionally has no image-path argument.
- Treat a request that preserves, combines, or changes an existing visual as an edit and call `ImageEdit`.
- One distinct prompt equals one tool call.
- Use `count` only for multiple variations of the same prompt. For different concepts, make separate calls.
- For an edit, populate `input_images` with ordered paths surfaced by `[Image source: ...]` in a user attachment or returned by an earlier `ImageGen` call. Never invent, search for, or substitute another filesystem path.
- Label every input image with its role: `edit_target`, `reference`, `style_reference`, or `composite_source`. The first image is the primary canvas unless the user says otherwise.
- `ImageEdit` requires `referenced_image_paths`: populate it with ordered, exact paths surfaced by `[Image source: ...]` in a user attachment or returned by an earlier `ImageGen` call. Never invent, search for, or substitute another filesystem path. The first image is the primary canvas unless the user says otherwise.
- For multi-turn editing, use the latest selected output as the next turn's `edit_target`. Repeat all identity, layout, text, and unchanged-region constraints on every turn so edits do not drift.
- To edit several images independently, make one call per image. Put multiple images in one call only when the user wants them combined or used together as references. A single call accepts at most three source images.
- Prefer a useful default composition when the user leaves details open. Do not invent branding, logos, or people they did not request.
- Provider and image model selection come from the current desktop session; do not add either to the tool arguments.
- If the provider or tool returns an error, do not retry `ImageGen` automatically. Explain the failure and let the user decide whether to retry or change providers.
- If the provider returns an error, do not retry the image tool automatically. Explain the failure and let the user decide whether to retry or change providers.
## Build the prompt
Turn the request into a compact art-direction brief. Include only relevant fields:
Turn the request into a complete art-direction brief. Preserve all relevant user-specified detail.
- use case and image type
- subject, action, and important attributes
+2 -1
View File
@@ -11,7 +11,7 @@ import { NotebookEditTool } from './tools/NotebookEditTool/NotebookEditTool.js'
import { WebFetchTool } from './tools/WebFetchTool/WebFetchTool.js'
import { TaskStopTool } from './tools/TaskStopTool/TaskStopTool.js'
import { BriefTool } from './tools/BriefTool/BriefTool.js'
import { ImageGenTool } from './tools/ImageGenTool/ImageGenTool.js'
import { ImageEditTool, ImageGenTool } from './tools/ImageGenTool/ImageGenTool.js'
// Dead code elimination: conditional import for ant-only tools
/* eslint-disable custom-rules/no-process-env-top-level, @typescript-eslint/no-require-imports */
const REPLTool =
@@ -243,6 +243,7 @@ export function getAllBaseTools(): Tools {
...(MonitorTool ? [MonitorTool] : []),
BriefTool,
ImageGenTool,
ImageEditTool,
...(SendUserFileTool ? [SendUserFileTool] : []),
...(PushNotificationTool ? [PushNotificationTool] : []),
...(SubscribePRTool ? [SubscribePRTool] : []),
+99 -10
View File
@@ -8,7 +8,10 @@ import {
IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY,
} from '../../services/imageGeneration/config.js'
import { zodToJsonSchema } from '../../utils/zodToJsonSchema.js'
import { ImageGenTool } from './ImageGenTool.js'
import {
ImageEditTool,
ImageGenTool,
} from './ImageGenTool.js'
const ENV_KEYS = [
IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY,
@@ -22,6 +25,19 @@ const originalEnv = Object.fromEntries(
ENV_KEYS.map((key) => [key, process.env[key]]),
)
function setImageRuntime(kind: 'openai_oauth' | 'grok_oauth' | 'openai_images') {
process.env[IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY] = kind
process.env[IMAGE_GENERATION_PROVIDER_ID_ENV_KEY] = `${kind}-provider`
process.env[IMAGE_GENERATION_MODEL_ENV_KEY] = `${kind}-image-model`
if (kind === 'openai_images') {
process.env[IMAGE_GENERATION_BASE_URL_ENV_KEY] = 'https://relay.test/v1'
process.env[IMAGE_GENERATION_API_KEY_ENV_KEY] = 'relay-key'
} else {
delete process.env[IMAGE_GENERATION_BASE_URL_ENV_KEY]
delete process.env[IMAGE_GENERATION_API_KEY_ENV_KEY]
}
}
afterEach(() => {
for (const key of ENV_KEYS) {
const value = originalEnv[key]
@@ -46,10 +62,14 @@ describe('ImageGenTool', () => {
})
test('defaults to one slot and preserves the structured result for the desktop', () => {
setImageRuntime('openai_images')
expect(ImageGenTool.strict).toBe(false)
const jsonSchema = zodToJsonSchema(ImageGenTool.inputSchema)
expect(jsonSchema.required).toEqual(['prompt', 'count'])
expect(jsonSchema.properties).not.toHaveProperty('model')
expect(jsonSchema.properties).not.toHaveProperty('input_images')
expect(jsonSchema.properties).not.toHaveProperty('referenced_image_paths')
expect(jsonSchema.properties?.prompt).not.toHaveProperty('maxLength')
expect(ImageGenTool.inputSchema.parse({ prompt: 'A paper-cut fox' })).toEqual({
prompt: 'A paper-cut fox',
count: 1,
@@ -58,6 +78,9 @@ describe('ImageGenTool', () => {
prompt: 'A paper-cut fox',
model: 'default',
})).toThrow()
expect(ImageGenTool.inputSchema.parse({
prompt: 'x'.repeat(8001),
}).prompt).toHaveLength(8001)
const output = {
type: 'image_generation_result' as const,
@@ -82,30 +105,96 @@ describe('ImageGenTool', () => {
})
})
test('accepts ordered edit targets and reference images', () => {
expect(ImageGenTool.inputSchema.parse({
test('separates editing into a schema that requires real source paths', () => {
setImageRuntime('openai_oauth')
const jsonSchema = zodToJsonSchema(ImageEditTool.inputSchema)
expect(jsonSchema.required).toEqual([
'prompt',
'count',
'referenced_image_paths',
])
expect(() => ImageEditTool.inputSchema.parse({
prompt: 'Change the scarf color',
})).toThrow()
expect(ImageEditTool.inputSchema.parse({
prompt: 'Combine these subjects while preserving their identity',
input_images: [
{ path: '/staged/first.png', role: 'edit_target' },
{ path: '/staged/second.png', role: 'composite_source' },
referenced_image_paths: [
'/staged/first.png',
'/staged/second.png',
],
})).toEqual({
prompt: 'Combine these subjects while preserving their identity',
count: 1,
input_images: [
{ path: '/staged/first.png', role: 'edit_target' },
{ path: '/staged/second.png', role: 'composite_source' },
referenced_image_paths: [
'/staged/first.png',
'/staged/second.png',
],
})
})
test('exposes image editing as a first-class tool without generation placeholders', async () => {
setImageRuntime('openai_oauth')
const input = ImageEditTool.inputSchema.parse({
prompt: 'Change only the scarf color',
count: 2,
referenced_image_paths: ['/staged/fox.png'],
})
const output = {
type: 'image_generation_result' as const,
operation: 'edit' as const,
inputImageCount: 1,
providerId: 'openai-official',
providerKind: 'openai_oauth' as const,
model: 'gpt-image-2',
prompt: input.prompt,
images: [{ path: '/tmp/edited.png', mimeType: 'image/png' as const }],
durationMs: 42,
}
expect(await ImageEditTool.description()).toContain('Edit images using exact source paths')
expect(ImageEditTool.outputSchema.parse(output)).toEqual(output)
expect(ImageEditTool.isEnabled()).toBe(true)
expect(ImageEditTool.isConcurrencySafe()).toBe(true)
expect(ImageEditTool.isReadOnly()).toBe(false)
expect(ImageEditTool.toAutoClassifierInput(input)).toBe(
'2 image edit(s): Change only the scarf color',
)
await expect(ImageEditTool.checkPermissions(input)).resolves.toEqual({
behavior: 'allow',
updatedInput: input,
})
expect(ImageEditTool.getToolUseSummary(input)).toBe(input.prompt)
expect(ImageEditTool.getToolUseSummary({ ...input, prompt: ' ' })).toBeNull()
expect(ImageEditTool.getActivityDescription(input)).toBe('Editing 2 image variations')
expect(ImageEditTool.getActivityDescription({ ...input, count: 1 })).toBe('Editing image')
expect(ImageEditTool.renderToolUseMessage()).toBeNull()
expect(ImageEditTool.mapToolResultToToolResultBlockParam(output, 'edit-use')).toEqual({
tool_use_id: 'edit-use',
type: 'tool_result',
content: JSON.stringify(output),
})
for (const key of ENV_KEYS) delete process.env[key]
await expect(ImageEditTool.call(input, {
abortController: new AbortController(),
} as never)).rejects.toThrow('Image generation is not configured')
})
test('tells the agent not to retry provider failures automatically', async () => {
setImageRuntime('openai_oauth')
expect(await ImageGenTool.description()).toContain('Generate brand-new images')
const prompt = await ImageGenTool.prompt()
expect(prompt).toContain(
'do not retry ImageGen automatically',
)
expect(prompt).toContain('omit input_images entirely')
expect(prompt).toContain('has no source-image argument')
expect(prompt).toContain('Preserve the full relevant user specification')
expect(prompt).toContain('Provider and image model selection')
expect(prompt).toContain('are not tool arguments')
const editPrompt = await ImageEditTool.prompt()
expect(editPrompt).toContain('referenced_image_paths is required')
expect(editPrompt).toContain('never invent, search for, or substitute a path')
expect(editPrompt).toContain('do not retry ImageEdit automatically')
})
})
+98 -29
View File
@@ -9,7 +9,10 @@ import {
generateImages,
type ImageGenerationOutput,
} from './backend.js'
import { IMAGE_GEN_TOOL_NAME } from './constants.js'
import {
IMAGE_EDIT_TOOL_NAME,
IMAGE_GEN_TOOL_NAME,
} from './constants.js'
const ASPECT_RATIOS = [
'auto',
@@ -28,12 +31,12 @@ const ASPECT_RATIOS = [
'9:20',
] as const
const inputSchema = lazySchema(() =>
z.strictObject({
function commonInputShape() {
return {
prompt: z
.string()
.min(1)
.describe('A complete visual prompt describing the image to generate'),
.describe('A complete visual prompt preserving all relevant user-specified detail'),
count: z
.number()
.int()
@@ -41,20 +44,6 @@ const inputSchema = lazySchema(() =>
.max(4)
.default(1)
.describe('Number of variations for this exact prompt, from 1 to 4'),
input_images: z
.array(z.strictObject({
path: z
.string()
.min(1)
.describe('Absolute path from an [Image source: ...] attachment or a prior ImageGen result; omit input_images entirely for new generation'),
role: z
.enum(['edit_target', 'reference', 'style_reference', 'composite_source'])
.describe('How this ordered image should influence the edit'),
}))
.min(1)
.max(3)
.optional()
.describe('Ordered source images for editing, compositing, or visual reference'),
aspect_ratio: z
.enum(ASPECT_RATIOS)
.optional()
@@ -70,9 +59,28 @@ const inputSchema = lazySchema(() =>
quality: z.enum(['auto', 'low', 'medium', 'high']).optional(),
background: z.enum(['auto', 'opaque', 'transparent']).optional(),
output_format: z.enum(['png', 'jpeg', 'webp']).optional(),
}
}
const generationInputSchema = lazySchema(() =>
z.strictObject(commonInputShape()),
)
type GenerationInputSchema = ReturnType<typeof generationInputSchema>
const editInputSchema = lazySchema(() =>
z.strictObject({
...commonInputShape(),
referenced_image_paths: z
.array(z
.string()
.min(1)
.describe('Exact absolute path from an [Image source: ...] attachment or a prior ImageGen result'))
.min(1)
.max(3)
.describe('Ordered source images to edit or use as visual references'),
}),
)
type InputSchema = ReturnType<typeof inputSchema>
type EditInputSchema = ReturnType<typeof editInputSchema>
const generatedImageSchema = z.object({
path: z.string(),
@@ -102,13 +110,13 @@ export const ImageGenTool = buildTool({
strict: false,
shouldDefer: true,
async description() {
return 'Generate one or more images with the image provider configured for this desktop session.'
return 'Generate brand-new images from a text prompt with the image provider configured for this desktop session. This tool does not accept source-image paths; use ImageEdit for edits.'
},
async prompt() {
return `Use this tool when the user asks to generate or edit an image. For a brand-new image, omit input_images entirely. For edits, pass ordered input_images using only paths surfaced by [Image source: ...] in the current conversation or returned by a prior ImageGen call; repeat preservation constraints in every edit prompt. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same prompt. The tool saves finished raster images locally and returns their absolute paths. If a provider call fails, do not retry ImageGen automatically; explain the error and wait for the user to decide.`
return 'Use this tool only for a brand-new image. It has no source-image argument; never invent or pass a placeholder path. Preserve the full relevant user specification. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same prompt. The tool saves finished raster images locally and returns their absolute paths. If a provider call fails, do not retry ImageGen automatically; explain the error and wait for the user to decide.'
},
get inputSchema(): InputSchema {
return inputSchema()
get inputSchema(): GenerationInputSchema {
return generationInputSchema()
},
get outputSchema(): OutputSchema {
return outputSchema()
@@ -132,11 +140,6 @@ export const ImageGenTool = buildTool({
return input?.prompt?.trim() || null
},
getActivityDescription(input) {
if (input?.input_images?.length) {
return input.count && input.count > 1
? `Editing ${input.count} image variations`
: 'Editing image'
}
return input?.count && input.count > 1
? `Generating ${input.count} images`
: 'Generating image'
@@ -164,4 +167,70 @@ export const ImageGenTool = buildTool({
content: JSON.stringify(output),
}
},
} satisfies ToolDef<InputSchema, ImageGenerationOutput>)
} satisfies ToolDef<GenerationInputSchema, ImageGenerationOutput>)
export const ImageEditTool = buildTool({
name: IMAGE_EDIT_TOOL_NAME,
searchHint: 'edit or transform an attached or previously generated image',
maxResultSizeChars: 100_000,
strict: false,
shouldDefer: true,
async description() {
return 'Edit images using exact source paths from user attachments or earlier ImageGen results.'
},
async prompt() {
return 'Use this tool only when the user wants to edit, combine, or visually reference existing images. referenced_image_paths is required and may contain only exact paths surfaced by [Image source: ...] in the current conversation or returned by a prior ImageGen call; never invent, search for, or substitute a path. Preserve the full relevant user specification and repeat preservation constraints in every edit prompt. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same edit. If a provider call fails, do not retry ImageEdit automatically; explain the error and wait for the user to decide.'
},
get inputSchema(): EditInputSchema {
return editInputSchema()
},
get outputSchema(): OutputSchema {
return outputSchema()
},
isEnabled() {
return getImageGenerationRuntimeConfig() !== null
},
isConcurrencySafe() {
return true
},
isReadOnly() {
return false
},
toAutoClassifierInput(input) {
return `${input.count} image edit(s): ${input.prompt}`
},
async checkPermissions(input) {
return { behavior: 'allow', updatedInput: input }
},
getToolUseSummary(input) {
return input?.prompt?.trim() || null
},
getActivityDescription(input) {
return input?.count && input.count > 1
? `Editing ${input.count} image variations`
: 'Editing image'
},
renderToolUseMessage() {
return null
},
async call(input, context) {
const config = getImageGenerationRuntimeConfig()
if (!config) {
throw new Error(
'Image generation is not configured for the current provider. Enable it in provider settings.',
)
}
return {
data: await generateImages(input, config, {
signal: context.abortController.signal,
}),
}
},
mapToolResultToToolResultBlockParam(output, toolUseID) {
return {
tool_use_id: toolUseID,
type: 'tool_result',
content: JSON.stringify(output),
}
},
} satisfies ToolDef<EditInputSchema, ImageGenerationOutput>)
+8 -8
View File
@@ -72,9 +72,9 @@ describe('ImageGen backend', () => {
prompt: 'Put a red scarf on the fox; keep everything else unchanged',
count: 1,
model: 'gpt-image-2',
input_images: [
{ path: '/allowed/fox.png', role: 'edit_target' },
{ path: '/allowed/scarf.jpg', role: 'reference' },
referenced_image_paths: [
'/allowed/fox.png',
'/allowed/scarf.jpg',
],
}, [
{ dataUrl: 'data:image/png;base64,Zm94', fileName: 'fox.png', mimeType: 'image/png', bytes: Buffer.from('fox') },
@@ -125,9 +125,9 @@ describe('ImageGen backend', () => {
prompt: 'Place both subjects together',
count: 1,
model: 'grok-imagine-image-quality',
input_images: [
{ path: '/allowed/first.png', role: 'edit_target' },
{ path: '/allowed/second.png', role: 'composite_source' },
referenced_image_paths: [
'/allowed/first.png',
'/allowed/second.png',
],
aspect_ratio: '3:2',
}, [
@@ -222,7 +222,7 @@ describe('ImageGen backend', () => {
const result = await generateImages({
prompt: 'Make the sky purple; preserve the subject and framing',
count: 2,
input_images: [{ path: sourcePath, role: 'edit_target' }],
referenced_image_paths: [sourcePath],
quality: 'high',
}, customConfig, {
fetchImpl,
@@ -308,7 +308,7 @@ describe('ImageGen backend', () => {
await expect(generateImages({
prompt: 'Edit this image',
count: 1,
input_images: [{ path: outsidePath, role: 'edit_target' }],
referenced_image_paths: [outsidePath],
}, customConfig, {
outputDir,
inputRootDirs: [outputDir],
+3 -6
View File
@@ -26,10 +26,7 @@ import { getProxyFetchOptions } from '../../utils/proxy.js'
export type ImageGenerationInput = {
prompt: string
count: number
input_images?: Array<{
path: string
role: 'edit_target' | 'reference' | 'style_reference' | 'composite_source'
}>
referenced_image_paths?: string[]
aspect_ratio?: string
resolution?: '1k' | '2k'
size?: 'auto' | '1024x1024' | '1024x1536' | '1536x1024'
@@ -606,7 +603,7 @@ async function prepareInputImages(
input: ImageGenerationInput,
overrideRootDirs?: string[],
): Promise<PreparedInputImage[]> {
const requested = input.input_images ?? []
const requested = input.referenced_image_paths ?? []
if (requested.length === 0) return []
if (requested.length > 3) {
throw new Error('Image editing supports at most 3 input images per call.')
@@ -617,7 +614,7 @@ async function prepareInputImages(
rootDirs.map(async (rootDir) => realpath(rootDir).catch(() => resolve(rootDir))),
)
return Promise.all(requested.map(async ({ path: inputPath }) => {
return Promise.all(requested.map(async (inputPath) => {
const resolvedPath = await realpath(inputPath).catch(() => null)
if (
!resolvedPath ||
+1
View File
@@ -1 +1,2 @@
export const IMAGE_GEN_TOOL_NAME = 'ImageGen'
export const IMAGE_EDIT_TOOL_NAME = 'ImageEdit'