From ba5dcffb222c86830ef0f6e46f5276e2de25d389 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F=28Relakkes?= =?UTF-8?q?=29?= Date: Wed, 5 Aug 2026 03:30:50 +0800 Subject: [PATCH] fix(imagegen): separate generation and editing tools --- .../components/chat/ImageGenerationBlock.tsx | 3 + desktop/src/components/chat/ToolCallBlock.tsx | 3 +- desktop/src/components/chat/ToolCallGroup.tsx | 9 +- .../src/components/chat/chatBlocks.test.tsx | 45 ++++++- .../components/chat/imageGenerationTools.ts | 3 + src/skills/bundled/imagegen.test.ts | 11 +- src/skills/bundled/imagegen.ts | 2 +- src/skills/bundled/imagegen/SKILL.md | 15 +-- src/tools.ts | 3 +- src/tools/ImageGenTool/ImageGenTool.test.ts | 109 +++++++++++++-- src/tools/ImageGenTool/ImageGenTool.ts | 127 ++++++++++++++---- src/tools/ImageGenTool/backend.test.ts | 16 +-- src/tools/ImageGenTool/backend.ts | 9 +- src/tools/ImageGenTool/constants.ts | 1 + 14 files changed, 281 insertions(+), 75 deletions(-) create mode 100644 desktop/src/components/chat/imageGenerationTools.ts diff --git a/desktop/src/components/chat/ImageGenerationBlock.tsx b/desktop/src/components/chat/ImageGenerationBlock.tsx index 57c80bde..4ba63e6b 100644 --- a/desktop/src/components/chat/ImageGenerationBlock.tsx +++ b/desktop/src/components/chat/ImageGenerationBlock.tsx @@ -48,6 +48,9 @@ export function ImageGenerationBlock({ const slotCount = requestedCount ?? parsedResult?.images.length ?? 1 const prompt = stringValue(inputRecord.prompt) ?? parsedResult?.prompt ?? '' const isEdit = ( + Array.isArray(inputRecord.referenced_image_paths) && inputRecord.referenced_image_paths.length > 0 + ) || ( + // Persisted calls from before the official imagegen argument alignment. Array.isArray(inputRecord.input_images) && inputRecord.input_images.length > 0 ) || parsedResult?.operation === 'edit' const isWaiting = !result diff --git a/desktop/src/components/chat/ToolCallBlock.tsx b/desktop/src/components/chat/ToolCallBlock.tsx index 62089bcf..e1f9bf1b 100644 --- a/desktop/src/components/chat/ToolCallBlock.tsx +++ b/desktop/src/components/chat/ToolCallBlock.tsx @@ -8,6 +8,7 @@ import { useTranslation } from '../../i18n' import type { TranslationKey } from '../../i18n' import { InlineImageGallery } from './InlineImageGallery' import { ImageGenerationBlock } from './ImageGenerationBlock' +import { isImageGenerationToolName } from './imageGenerationTools' import type { AgentTaskNotification } from '../../types/chat' import { PlanPreviewCard, @@ -187,7 +188,7 @@ export const ToolCallBlock = memo(function ToolCallBlock({ toolName, input, resu ) } - if (toolName === 'ImageGen') { + if (isImageGenerationToolName(toolName)) { return ( toolCall.toolName === 'ImageGen') - if (hasImageGeneration && !toolCalls.every((toolCall) => toolCall.toolName === 'ImageGen')) { + const hasImageGeneration = toolCalls.some((toolCall) => isImageGenerationToolName(toolCall.toolName)) + if (hasImageGeneration && !toolCalls.every((toolCall) => isImageGenerationToolName(toolCall.toolName))) { const segments: Array< | { kind: 'image'; toolCall: ToolCall } | { kind: 'regular'; toolCalls: ToolCall[] } @@ -217,7 +218,7 @@ function ToolCallGroupContent({ } for (const toolCall of toolCalls) { - if (toolCall.toolName === 'ImageGen') { + if (isImageGenerationToolName(toolCall.toolName)) { flushRegularCalls() segments.push({ kind: 'image', toolCall }) } else { @@ -268,7 +269,7 @@ function ToolCallGroupContent({ ) } - const allImageGeneration = toolCalls.length > 0 && toolCalls.every((toolCall) => toolCall.toolName === 'ImageGen') + const allImageGeneration = toolCalls.length > 0 && toolCalls.every((toolCall) => isImageGenerationToolName(toolCall.toolName)) if (allImageGeneration) { return (
diff --git a/desktop/src/components/chat/chatBlocks.test.tsx b/desktop/src/components/chat/chatBlocks.test.tsx index 6fc03f2e..3cf84036 100644 --- a/desktop/src/components/chat/chatBlocks.test.tsx +++ b/desktop/src/components/chat/chatBlocks.test.tsx @@ -47,14 +47,14 @@ describe('chat blocks', () => { expect(screen.getAllByText('Generating 2 images')).toHaveLength(2) }) - it('labels an input-image turn as editing while keeping output placeholders', () => { + it('labels a referenced-image turn as editing while keeping output placeholders', () => { render( , @@ -99,6 +99,45 @@ describe('chat blocks', () => { expect(screen.queryByRole('button', { name: /ToolSearch \(1\), ImageGen \(1\)/ })).toBeNull() }) + it('keeps image editing outside a mixed deferred-tool summary', () => { + const toolCalls: Array> = [ + { + id: 'search-use', + type: 'tool_use', + toolName: 'ToolSearch', + toolUseId: 'search-1', + input: { query: 'image editing' }, + timestamp: 1, + }, + { + id: 'edit-use', + type: 'tool_use', + toolName: 'ImageEdit', + toolUseId: 'edit-1', + input: { + prompt: 'Change only the scarf color', + count: 1, + referenced_image_paths: ['/staged/fox.png'], + }, + timestamp: 2, + isPending: true, + }, + ] + + render( + , + ) + + expect(screen.getAllByTestId('image-generation-slot')).toHaveLength(1) + expect(screen.queryByRole('button', { name: /ToolSearch \(1\), ImageEdit \(1\)/ })).toBeNull() + }) + it('replaces every image placeholder with the saved tool result', () => { const content = JSON.stringify({ type: 'image_generation_result', diff --git a/desktop/src/components/chat/imageGenerationTools.ts b/desktop/src/components/chat/imageGenerationTools.ts new file mode 100644 index 00000000..f49b17c2 --- /dev/null +++ b/desktop/src/components/chat/imageGenerationTools.ts @@ -0,0 +1,3 @@ +export function isImageGenerationToolName(toolName: string): boolean { + return toolName === 'ImageGen' || toolName === 'ImageEdit' +} diff --git a/src/skills/bundled/imagegen.test.ts b/src/skills/bundled/imagegen.test.ts index 107efe09..1decb2af 100644 --- a/src/skills/bundled/imagegen.test.ts +++ b/src/skills/bundled/imagegen.test.ts @@ -39,7 +39,7 @@ describe('bundled imagegen skill', () => { const skill = getBundledSkills().find((command) => command.name === 'imagegen') expect(skill).toBeDefined() - expect(skill?.allowedTools).toEqual(['ImageGen']) + expect(skill?.allowedTools).toEqual(['ImageGen', 'ImageEdit']) expect(skill?.isEnabled?.()).toBe(true) if (!skill || skill.type !== 'prompt') return @@ -53,12 +53,15 @@ describe('bundled imagegen skill', () => { .join('\n') expect(text).toContain('Create two poster variants') expect(text).toContain('ImageGen') - expect(text).toContain('do not retry ImageGen automatically') + expect(text).toContain('ImageEdit') + expect(text).toContain('do not retry the image tool automatically') expect(text).toContain('do not repeat, link, or embed the returned local paths') expect(text).toContain('latest selected output as the next turn') expect(text).toContain('one call per image') - expect(text).toContain('omit input_images entirely') - expect(text).toContain('Never invent a path') + expect(text).toContain('schema intentionally has no image-path argument') + expect(text).toContain('ImageEdit requires referenced_image_paths') + expect(text).toContain('Never invent, search for, or substitute another filesystem path') + expect(text).toContain('Preserve all relevant user-specified detail') expect(text).toContain('Provider and image model selection come from') expect(text).toContain('do not add either to the tool arguments') expect(text).not.toContain('CC_HAHA_IMAGE_API_KEY') diff --git a/src/skills/bundled/imagegen.ts b/src/skills/bundled/imagegen.ts index 93b9856f..e7639886 100644 --- a/src/skills/bundled/imagegen.ts +++ b/src/skills/bundled/imagegen.ts @@ -14,7 +14,7 @@ export function registerImagegenSkill(): void { registerBundledSkill({ name: 'imagegen', description: DESCRIPTION, - allowedTools: ['ImageGen'], + allowedTools: ['ImageGen', 'ImageEdit'], userInvocable: true, isEnabled: () => getImageGenerationRuntimeConfig() !== null, async getPromptForCommand(args) { diff --git a/src/skills/bundled/imagegen/SKILL.md b/src/skills/bundled/imagegen/SKILL.md index 0965f5dc..fc4d9706 100644 --- a/src/skills/bundled/imagegen/SKILL.md +++ b/src/skills/bundled/imagegen/SKILL.md @@ -1,30 +1,29 @@ --- name: imagegen description: Generate original images, artwork, product visuals, diagrams, or other raster assets with the desktop's configured image provider. Use whenever the user asks to create or generate an image. -allowed-tools: ImageGen +allowed-tools: ImageGen, ImageEdit --- # Image generation -Use the built-in `ImageGen` tool. Provider authentication, model routing, output storage, and secrets are managed by the desktop host; never ask the user to put an API key in this skill or in the prompt. +Use the built-in `ImageGen` and `ImageEdit` tools. Provider authentication, model routing, output storage, and secrets are managed by the desktop host; never ask the user to put an API key in this skill or in the prompt. ## Decide the request shape -- Treat a brand-new visual as generation. Treat a request that preserves or changes an existing visual as an edit. -- For generation, omit `input_images` entirely. Never invent a path to represent the absence of an input image. +- Treat a brand-new visual as generation and call `ImageGen`. Its schema intentionally has no image-path argument. +- Treat a request that preserves, combines, or changes an existing visual as an edit and call `ImageEdit`. - One distinct prompt equals one tool call. - Use `count` only for multiple variations of the same prompt. For different concepts, make separate calls. -- For an edit, populate `input_images` with ordered paths surfaced by `[Image source: ...]` in a user attachment or returned by an earlier `ImageGen` call. Never invent, search for, or substitute another filesystem path. -- Label every input image with its role: `edit_target`, `reference`, `style_reference`, or `composite_source`. The first image is the primary canvas unless the user says otherwise. +- `ImageEdit` requires `referenced_image_paths`: populate it with ordered, exact paths surfaced by `[Image source: ...]` in a user attachment or returned by an earlier `ImageGen` call. Never invent, search for, or substitute another filesystem path. The first image is the primary canvas unless the user says otherwise. - For multi-turn editing, use the latest selected output as the next turn's `edit_target`. Repeat all identity, layout, text, and unchanged-region constraints on every turn so edits do not drift. - To edit several images independently, make one call per image. Put multiple images in one call only when the user wants them combined or used together as references. A single call accepts at most three source images. - Prefer a useful default composition when the user leaves details open. Do not invent branding, logos, or people they did not request. - Provider and image model selection come from the current desktop session; do not add either to the tool arguments. -- If the provider or tool returns an error, do not retry `ImageGen` automatically. Explain the failure and let the user decide whether to retry or change providers. +- If the provider returns an error, do not retry the image tool automatically. Explain the failure and let the user decide whether to retry or change providers. ## Build the prompt -Turn the request into a compact art-direction brief. Include only relevant fields: +Turn the request into a complete art-direction brief. Preserve all relevant user-specified detail. - use case and image type - subject, action, and important attributes diff --git a/src/tools.ts b/src/tools.ts index 5e24b0ff..3e4e23d0 100644 --- a/src/tools.ts +++ b/src/tools.ts @@ -11,7 +11,7 @@ import { NotebookEditTool } from './tools/NotebookEditTool/NotebookEditTool.js' import { WebFetchTool } from './tools/WebFetchTool/WebFetchTool.js' import { TaskStopTool } from './tools/TaskStopTool/TaskStopTool.js' import { BriefTool } from './tools/BriefTool/BriefTool.js' -import { ImageGenTool } from './tools/ImageGenTool/ImageGenTool.js' +import { ImageEditTool, ImageGenTool } from './tools/ImageGenTool/ImageGenTool.js' // Dead code elimination: conditional import for ant-only tools /* eslint-disable custom-rules/no-process-env-top-level, @typescript-eslint/no-require-imports */ const REPLTool = @@ -243,6 +243,7 @@ export function getAllBaseTools(): Tools { ...(MonitorTool ? [MonitorTool] : []), BriefTool, ImageGenTool, + ImageEditTool, ...(SendUserFileTool ? [SendUserFileTool] : []), ...(PushNotificationTool ? [PushNotificationTool] : []), ...(SubscribePRTool ? [SubscribePRTool] : []), diff --git a/src/tools/ImageGenTool/ImageGenTool.test.ts b/src/tools/ImageGenTool/ImageGenTool.test.ts index 45968cdb..82e3731a 100644 --- a/src/tools/ImageGenTool/ImageGenTool.test.ts +++ b/src/tools/ImageGenTool/ImageGenTool.test.ts @@ -8,7 +8,10 @@ import { IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY, } from '../../services/imageGeneration/config.js' import { zodToJsonSchema } from '../../utils/zodToJsonSchema.js' -import { ImageGenTool } from './ImageGenTool.js' +import { + ImageEditTool, + ImageGenTool, +} from './ImageGenTool.js' const ENV_KEYS = [ IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY, @@ -22,6 +25,19 @@ const originalEnv = Object.fromEntries( ENV_KEYS.map((key) => [key, process.env[key]]), ) +function setImageRuntime(kind: 'openai_oauth' | 'grok_oauth' | 'openai_images') { + process.env[IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY] = kind + process.env[IMAGE_GENERATION_PROVIDER_ID_ENV_KEY] = `${kind}-provider` + process.env[IMAGE_GENERATION_MODEL_ENV_KEY] = `${kind}-image-model` + if (kind === 'openai_images') { + process.env[IMAGE_GENERATION_BASE_URL_ENV_KEY] = 'https://relay.test/v1' + process.env[IMAGE_GENERATION_API_KEY_ENV_KEY] = 'relay-key' + } else { + delete process.env[IMAGE_GENERATION_BASE_URL_ENV_KEY] + delete process.env[IMAGE_GENERATION_API_KEY_ENV_KEY] + } +} + afterEach(() => { for (const key of ENV_KEYS) { const value = originalEnv[key] @@ -46,10 +62,14 @@ describe('ImageGenTool', () => { }) test('defaults to one slot and preserves the structured result for the desktop', () => { + setImageRuntime('openai_images') expect(ImageGenTool.strict).toBe(false) const jsonSchema = zodToJsonSchema(ImageGenTool.inputSchema) expect(jsonSchema.required).toEqual(['prompt', 'count']) expect(jsonSchema.properties).not.toHaveProperty('model') + expect(jsonSchema.properties).not.toHaveProperty('input_images') + expect(jsonSchema.properties).not.toHaveProperty('referenced_image_paths') + expect(jsonSchema.properties?.prompt).not.toHaveProperty('maxLength') expect(ImageGenTool.inputSchema.parse({ prompt: 'A paper-cut fox' })).toEqual({ prompt: 'A paper-cut fox', count: 1, @@ -58,6 +78,9 @@ describe('ImageGenTool', () => { prompt: 'A paper-cut fox', model: 'default', })).toThrow() + expect(ImageGenTool.inputSchema.parse({ + prompt: 'x'.repeat(8001), + }).prompt).toHaveLength(8001) const output = { type: 'image_generation_result' as const, @@ -82,30 +105,96 @@ describe('ImageGenTool', () => { }) }) - test('accepts ordered edit targets and reference images', () => { - expect(ImageGenTool.inputSchema.parse({ + test('separates editing into a schema that requires real source paths', () => { + setImageRuntime('openai_oauth') + const jsonSchema = zodToJsonSchema(ImageEditTool.inputSchema) + expect(jsonSchema.required).toEqual([ + 'prompt', + 'count', + 'referenced_image_paths', + ]) + expect(() => ImageEditTool.inputSchema.parse({ + prompt: 'Change the scarf color', + })).toThrow() + expect(ImageEditTool.inputSchema.parse({ prompt: 'Combine these subjects while preserving their identity', - input_images: [ - { path: '/staged/first.png', role: 'edit_target' }, - { path: '/staged/second.png', role: 'composite_source' }, + referenced_image_paths: [ + '/staged/first.png', + '/staged/second.png', ], })).toEqual({ prompt: 'Combine these subjects while preserving their identity', count: 1, - input_images: [ - { path: '/staged/first.png', role: 'edit_target' }, - { path: '/staged/second.png', role: 'composite_source' }, + referenced_image_paths: [ + '/staged/first.png', + '/staged/second.png', ], }) }) + test('exposes image editing as a first-class tool without generation placeholders', async () => { + setImageRuntime('openai_oauth') + const input = ImageEditTool.inputSchema.parse({ + prompt: 'Change only the scarf color', + count: 2, + referenced_image_paths: ['/staged/fox.png'], + }) + const output = { + type: 'image_generation_result' as const, + operation: 'edit' as const, + inputImageCount: 1, + providerId: 'openai-official', + providerKind: 'openai_oauth' as const, + model: 'gpt-image-2', + prompt: input.prompt, + images: [{ path: '/tmp/edited.png', mimeType: 'image/png' as const }], + durationMs: 42, + } + + expect(await ImageEditTool.description()).toContain('Edit images using exact source paths') + expect(ImageEditTool.outputSchema.parse(output)).toEqual(output) + expect(ImageEditTool.isEnabled()).toBe(true) + expect(ImageEditTool.isConcurrencySafe()).toBe(true) + expect(ImageEditTool.isReadOnly()).toBe(false) + expect(ImageEditTool.toAutoClassifierInput(input)).toBe( + '2 image edit(s): Change only the scarf color', + ) + await expect(ImageEditTool.checkPermissions(input)).resolves.toEqual({ + behavior: 'allow', + updatedInput: input, + }) + expect(ImageEditTool.getToolUseSummary(input)).toBe(input.prompt) + expect(ImageEditTool.getToolUseSummary({ ...input, prompt: ' ' })).toBeNull() + expect(ImageEditTool.getActivityDescription(input)).toBe('Editing 2 image variations') + expect(ImageEditTool.getActivityDescription({ ...input, count: 1 })).toBe('Editing image') + expect(ImageEditTool.renderToolUseMessage()).toBeNull() + expect(ImageEditTool.mapToolResultToToolResultBlockParam(output, 'edit-use')).toEqual({ + tool_use_id: 'edit-use', + type: 'tool_result', + content: JSON.stringify(output), + }) + + for (const key of ENV_KEYS) delete process.env[key] + await expect(ImageEditTool.call(input, { + abortController: new AbortController(), + } as never)).rejects.toThrow('Image generation is not configured') + }) + test('tells the agent not to retry provider failures automatically', async () => { + setImageRuntime('openai_oauth') + expect(await ImageGenTool.description()).toContain('Generate brand-new images') const prompt = await ImageGenTool.prompt() expect(prompt).toContain( 'do not retry ImageGen automatically', ) - expect(prompt).toContain('omit input_images entirely') + expect(prompt).toContain('has no source-image argument') + expect(prompt).toContain('Preserve the full relevant user specification') expect(prompt).toContain('Provider and image model selection') expect(prompt).toContain('are not tool arguments') + + const editPrompt = await ImageEditTool.prompt() + expect(editPrompt).toContain('referenced_image_paths is required') + expect(editPrompt).toContain('never invent, search for, or substitute a path') + expect(editPrompt).toContain('do not retry ImageEdit automatically') }) }) diff --git a/src/tools/ImageGenTool/ImageGenTool.ts b/src/tools/ImageGenTool/ImageGenTool.ts index 97682a3e..ef0f8d30 100644 --- a/src/tools/ImageGenTool/ImageGenTool.ts +++ b/src/tools/ImageGenTool/ImageGenTool.ts @@ -9,7 +9,10 @@ import { generateImages, type ImageGenerationOutput, } from './backend.js' -import { IMAGE_GEN_TOOL_NAME } from './constants.js' +import { + IMAGE_EDIT_TOOL_NAME, + IMAGE_GEN_TOOL_NAME, +} from './constants.js' const ASPECT_RATIOS = [ 'auto', @@ -28,12 +31,12 @@ const ASPECT_RATIOS = [ '9:20', ] as const -const inputSchema = lazySchema(() => - z.strictObject({ +function commonInputShape() { + return { prompt: z .string() .min(1) - .describe('A complete visual prompt describing the image to generate'), + .describe('A complete visual prompt preserving all relevant user-specified detail'), count: z .number() .int() @@ -41,20 +44,6 @@ const inputSchema = lazySchema(() => .max(4) .default(1) .describe('Number of variations for this exact prompt, from 1 to 4'), - input_images: z - .array(z.strictObject({ - path: z - .string() - .min(1) - .describe('Absolute path from an [Image source: ...] attachment or a prior ImageGen result; omit input_images entirely for new generation'), - role: z - .enum(['edit_target', 'reference', 'style_reference', 'composite_source']) - .describe('How this ordered image should influence the edit'), - })) - .min(1) - .max(3) - .optional() - .describe('Ordered source images for editing, compositing, or visual reference'), aspect_ratio: z .enum(ASPECT_RATIOS) .optional() @@ -70,9 +59,28 @@ const inputSchema = lazySchema(() => quality: z.enum(['auto', 'low', 'medium', 'high']).optional(), background: z.enum(['auto', 'opaque', 'transparent']).optional(), output_format: z.enum(['png', 'jpeg', 'webp']).optional(), + } +} + +const generationInputSchema = lazySchema(() => + z.strictObject(commonInputShape()), +) +type GenerationInputSchema = ReturnType + +const editInputSchema = lazySchema(() => + z.strictObject({ + ...commonInputShape(), + referenced_image_paths: z + .array(z + .string() + .min(1) + .describe('Exact absolute path from an [Image source: ...] attachment or a prior ImageGen result')) + .min(1) + .max(3) + .describe('Ordered source images to edit or use as visual references'), }), ) -type InputSchema = ReturnType +type EditInputSchema = ReturnType const generatedImageSchema = z.object({ path: z.string(), @@ -102,13 +110,13 @@ export const ImageGenTool = buildTool({ strict: false, shouldDefer: true, async description() { - return 'Generate one or more images with the image provider configured for this desktop session.' + return 'Generate brand-new images from a text prompt with the image provider configured for this desktop session. This tool does not accept source-image paths; use ImageEdit for edits.' }, async prompt() { - return `Use this tool when the user asks to generate or edit an image. For a brand-new image, omit input_images entirely. For edits, pass ordered input_images using only paths surfaced by [Image source: ...] in the current conversation or returned by a prior ImageGen call; repeat preservation constraints in every edit prompt. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same prompt. The tool saves finished raster images locally and returns their absolute paths. If a provider call fails, do not retry ImageGen automatically; explain the error and wait for the user to decide.` + return 'Use this tool only for a brand-new image. It has no source-image argument; never invent or pass a placeholder path. Preserve the full relevant user specification. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same prompt. The tool saves finished raster images locally and returns their absolute paths. If a provider call fails, do not retry ImageGen automatically; explain the error and wait for the user to decide.' }, - get inputSchema(): InputSchema { - return inputSchema() + get inputSchema(): GenerationInputSchema { + return generationInputSchema() }, get outputSchema(): OutputSchema { return outputSchema() @@ -132,11 +140,6 @@ export const ImageGenTool = buildTool({ return input?.prompt?.trim() || null }, getActivityDescription(input) { - if (input?.input_images?.length) { - return input.count && input.count > 1 - ? `Editing ${input.count} image variations` - : 'Editing image' - } return input?.count && input.count > 1 ? `Generating ${input.count} images` : 'Generating image' @@ -164,4 +167,70 @@ export const ImageGenTool = buildTool({ content: JSON.stringify(output), } }, -} satisfies ToolDef) +} satisfies ToolDef) + +export const ImageEditTool = buildTool({ + name: IMAGE_EDIT_TOOL_NAME, + searchHint: 'edit or transform an attached or previously generated image', + maxResultSizeChars: 100_000, + strict: false, + shouldDefer: true, + async description() { + return 'Edit images using exact source paths from user attachments or earlier ImageGen results.' + }, + async prompt() { + return 'Use this tool only when the user wants to edit, combine, or visually reference existing images. referenced_image_paths is required and may contain only exact paths surfaced by [Image source: ...] in the current conversation or returned by a prior ImageGen call; never invent, search for, or substitute a path. Preserve the full relevant user specification and repeat preservation constraints in every edit prompt. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same edit. If a provider call fails, do not retry ImageEdit automatically; explain the error and wait for the user to decide.' + }, + get inputSchema(): EditInputSchema { + return editInputSchema() + }, + get outputSchema(): OutputSchema { + return outputSchema() + }, + isEnabled() { + return getImageGenerationRuntimeConfig() !== null + }, + isConcurrencySafe() { + return true + }, + isReadOnly() { + return false + }, + toAutoClassifierInput(input) { + return `${input.count} image edit(s): ${input.prompt}` + }, + async checkPermissions(input) { + return { behavior: 'allow', updatedInput: input } + }, + getToolUseSummary(input) { + return input?.prompt?.trim() || null + }, + getActivityDescription(input) { + return input?.count && input.count > 1 + ? `Editing ${input.count} image variations` + : 'Editing image' + }, + renderToolUseMessage() { + return null + }, + async call(input, context) { + const config = getImageGenerationRuntimeConfig() + if (!config) { + throw new Error( + 'Image generation is not configured for the current provider. Enable it in provider settings.', + ) + } + return { + data: await generateImages(input, config, { + signal: context.abortController.signal, + }), + } + }, + mapToolResultToToolResultBlockParam(output, toolUseID) { + return { + tool_use_id: toolUseID, + type: 'tool_result', + content: JSON.stringify(output), + } + }, +} satisfies ToolDef) diff --git a/src/tools/ImageGenTool/backend.test.ts b/src/tools/ImageGenTool/backend.test.ts index f6a5e22a..27b8685e 100644 --- a/src/tools/ImageGenTool/backend.test.ts +++ b/src/tools/ImageGenTool/backend.test.ts @@ -72,9 +72,9 @@ describe('ImageGen backend', () => { prompt: 'Put a red scarf on the fox; keep everything else unchanged', count: 1, model: 'gpt-image-2', - input_images: [ - { path: '/allowed/fox.png', role: 'edit_target' }, - { path: '/allowed/scarf.jpg', role: 'reference' }, + referenced_image_paths: [ + '/allowed/fox.png', + '/allowed/scarf.jpg', ], }, [ { dataUrl: 'data:image/png;base64,Zm94', fileName: 'fox.png', mimeType: 'image/png', bytes: Buffer.from('fox') }, @@ -125,9 +125,9 @@ describe('ImageGen backend', () => { prompt: 'Place both subjects together', count: 1, model: 'grok-imagine-image-quality', - input_images: [ - { path: '/allowed/first.png', role: 'edit_target' }, - { path: '/allowed/second.png', role: 'composite_source' }, + referenced_image_paths: [ + '/allowed/first.png', + '/allowed/second.png', ], aspect_ratio: '3:2', }, [ @@ -222,7 +222,7 @@ describe('ImageGen backend', () => { const result = await generateImages({ prompt: 'Make the sky purple; preserve the subject and framing', count: 2, - input_images: [{ path: sourcePath, role: 'edit_target' }], + referenced_image_paths: [sourcePath], quality: 'high', }, customConfig, { fetchImpl, @@ -308,7 +308,7 @@ describe('ImageGen backend', () => { await expect(generateImages({ prompt: 'Edit this image', count: 1, - input_images: [{ path: outsidePath, role: 'edit_target' }], + referenced_image_paths: [outsidePath], }, customConfig, { outputDir, inputRootDirs: [outputDir], diff --git a/src/tools/ImageGenTool/backend.ts b/src/tools/ImageGenTool/backend.ts index 2bbd884b..c4e517fa 100644 --- a/src/tools/ImageGenTool/backend.ts +++ b/src/tools/ImageGenTool/backend.ts @@ -26,10 +26,7 @@ import { getProxyFetchOptions } from '../../utils/proxy.js' export type ImageGenerationInput = { prompt: string count: number - input_images?: Array<{ - path: string - role: 'edit_target' | 'reference' | 'style_reference' | 'composite_source' - }> + referenced_image_paths?: string[] aspect_ratio?: string resolution?: '1k' | '2k' size?: 'auto' | '1024x1024' | '1024x1536' | '1536x1024' @@ -606,7 +603,7 @@ async function prepareInputImages( input: ImageGenerationInput, overrideRootDirs?: string[], ): Promise { - const requested = input.input_images ?? [] + const requested = input.referenced_image_paths ?? [] if (requested.length === 0) return [] if (requested.length > 3) { throw new Error('Image editing supports at most 3 input images per call.') @@ -617,7 +614,7 @@ async function prepareInputImages( rootDirs.map(async (rootDir) => realpath(rootDir).catch(() => resolve(rootDir))), ) - return Promise.all(requested.map(async ({ path: inputPath }) => { + return Promise.all(requested.map(async (inputPath) => { const resolvedPath = await realpath(inputPath).catch(() => null) if ( !resolvedPath || diff --git a/src/tools/ImageGenTool/constants.ts b/src/tools/ImageGenTool/constants.ts index b5e850ac..5237d172 100644 --- a/src/tools/ImageGenTool/constants.ts +++ b/src/tools/ImageGenTool/constants.ts @@ -1 +1,2 @@ export const IMAGE_GEN_TOOL_NAME = 'ImageGen' +export const IMAGE_EDIT_TOOL_NAME = 'ImageEdit'