fix(imagegen): align tool inputs with official contract

This commit is contained in:
程序员阿江(Relakkes)
2026-08-05 02:34:28 +08:00
parent 563003ec95
commit 8755a03f57
6 changed files with 22 additions and 71 deletions
+3 -3
View File
@@ -58,9 +58,9 @@ describe('bundled imagegen skill', () => {
expect(text).toContain('latest selected output as the next turn')
expect(text).toContain('one call per image')
expect(text).toContain('omit <code>input_images</code> entirely')
expect(text).toContain('Never pass <code>/dev/null</code>')
expect(text).toContain('otherwise omit the field')
expect(text).toContain('Never pass <code>default</code>')
expect(text).toContain('Never invent a path')
expect(text).toContain('Provider and image model selection come from')
expect(text).toContain('do not add either to the tool arguments')
expect(text).not.toContain('CC_HAHA_IMAGE_API_KEY')
})
})
+2 -2
View File
@@ -11,7 +11,7 @@ Use the built-in `ImageGen` tool. Provider authentication, model routing, output
## Decide the request shape
- Treat a brand-new visual as generation. Treat a request that preserves or changes an existing visual as an edit.
- For generation, omit `input_images` entirely. Never pass `/dev/null`, an empty string, or any other placeholder path to represent no input image.
- For generation, omit `input_images` entirely. Never invent a path to represent the absence of an input image.
- One distinct prompt equals one tool call.
- Use `count` only for multiple variations of the same prompt. For different concepts, make separate calls.
- For an edit, populate `input_images` with ordered paths surfaced by `[Image source: ...]` in a user attachment or returned by an earlier `ImageGen` call. Never invent, search for, or substitute another filesystem path.
@@ -19,7 +19,7 @@ Use the built-in `ImageGen` tool. Provider authentication, model routing, output
- For multi-turn editing, use the latest selected output as the next turn's `edit_target`. Repeat all identity, layout, text, and unchanged-region constraints on every turn so edits do not drift.
- To edit several images independently, make one call per image. Put multiple images in one call only when the user wants them combined or used together as references. A single call accepts at most three source images.
- Prefer a useful default composition when the user leaves details open. Do not invent branding, logos, or people they did not request.
- Respect an explicitly requested concrete provider model ID by passing `model`; otherwise omit the field so the configured model is used. Never pass `default` or another placeholder model name.
- Provider and image model selection come from the current desktop session; do not add either to the tool arguments.
- If the provider or tool returns an error, do not retry `ImageGen` automatically. Explain the failure and let the user decide whether to retry or change providers.
## Build the prompt
+11 -3
View File
@@ -7,6 +7,7 @@ import {
IMAGE_GENERATION_PROVIDER_ID_ENV_KEY,
IMAGE_GENERATION_PROVIDER_KIND_ENV_KEY,
} from '../../services/imageGeneration/config.js'
import { zodToJsonSchema } from '../../utils/zodToJsonSchema.js'
import { ImageGenTool } from './ImageGenTool.js'
const ENV_KEYS = [
@@ -45,10 +46,18 @@ describe('ImageGenTool', () => {
})
test('defaults to one slot and preserves the structured result for the desktop', () => {
expect(ImageGenTool.strict).toBe(false)
const jsonSchema = zodToJsonSchema(ImageGenTool.inputSchema)
expect(jsonSchema.required).toEqual(['prompt', 'count'])
expect(jsonSchema.properties).not.toHaveProperty('model')
expect(ImageGenTool.inputSchema.parse({ prompt: 'A paper-cut fox' })).toEqual({
prompt: 'A paper-cut fox',
count: 1,
})
expect(() => ImageGenTool.inputSchema.parse({
prompt: 'A paper-cut fox',
model: 'default',
})).toThrow()
const output = {
type: 'image_generation_result' as const,
@@ -96,8 +105,7 @@ describe('ImageGenTool', () => {
'do not retry ImageGen automatically',
)
expect(prompt).toContain('omit input_images entirely')
expect(prompt).toContain('never pass /dev/null')
expect(prompt).toContain('Omit model unless')
expect(prompt).toContain('never pass "default"')
expect(prompt).toContain('Provider and image model selection')
expect(prompt).toContain('are not tool arguments')
})
})
+3 -8
View File
@@ -46,7 +46,7 @@ const inputSchema = lazySchema(() =>
path: z
.string()
.min(1)
.describe('Absolute path from an [Image source: ...] attachment or a prior ImageGen result; omit input_images entirely for new generation and never use /dev/null as a placeholder'),
.describe('Absolute path from an [Image source: ...] attachment or a prior ImageGen result; omit input_images entirely for new generation'),
role: z
.enum(['edit_target', 'reference', 'style_reference', 'composite_source'])
.describe('How this ordered image should influence the edit'),
@@ -55,11 +55,6 @@ const inputSchema = lazySchema(() =>
.max(3)
.optional()
.describe('Ordered source images for editing, compositing, or visual reference'),
model: z
.string()
.min(1)
.optional()
.describe('Concrete image model ID to override the configured model only when the user explicitly asks; otherwise omit this field and never use "default" as a placeholder'),
aspect_ratio: z
.enum(ASPECT_RATIOS)
.optional()
@@ -104,13 +99,13 @@ export const ImageGenTool = buildTool({
name: IMAGE_GEN_TOOL_NAME,
searchHint: 'generate images or artwork from a visual prompt',
maxResultSizeChars: 100_000,
strict: true,
strict: false,
shouldDefer: true,
async description() {
return 'Generate one or more images with the image provider configured for this desktop session.'
},
async prompt() {
return `Use this tool when the user asks to generate or edit an image. For a brand-new image, omit input_images entirely; never pass /dev/null or another placeholder path. Omit model unless the user explicitly requests a concrete image model ID; never pass "default" as a placeholder. For edits, pass ordered input_images using only paths surfaced by [Image source: ...] in the current conversation or returned by a prior ImageGen call; repeat preservation constraints in every edit prompt. One call represents one distinct prompt; use count only for variations of that same prompt. The tool saves finished raster images locally and returns their absolute paths. If a provider call fails, do not retry ImageGen automatically; explain the error and wait for the user to decide.`
return `Use this tool when the user asks to generate or edit an image. For a brand-new image, omit input_images entirely. For edits, pass ordered input_images using only paths surfaced by [Image source: ...] in the current conversation or returned by a prior ImageGen call; repeat preservation constraints in every edit prompt. Provider and image model selection come from the current desktop session and are not tool arguments. One call represents one distinct prompt; use count only for variations of that same prompt. The tool saves finished raster images locally and returns their absolute paths. If a provider call fails, do not retry ImageGen automatically; explain the error and wait for the user to decide.`
},
get inputSchema(): InputSchema {
return inputSchema()
+1 -45
View File
@@ -246,50 +246,7 @@ describe('ImageGen backend', () => {
expect(result.images).toHaveLength(2)
})
test('treats a model-supplied /dev/null image placeholder as generation', async () => {
outputDir = await mkdtemp(join(tmpdir(), 'imagegen-output-'))
const calls: string[] = []
const fetchImpl = async (input: string | URL | Request) => {
calls.push(String(input))
return Response.json({
data: [{ b64_json: PNG_BYTES.toString('base64') }],
})
}
const result = await generateImages({
prompt: 'A paper-cut fox poster',
count: 1,
input_images: [{ path: '/dev/null', role: 'reference' }],
}, customConfig, { fetchImpl, outputDir })
expect(calls).toEqual([
'https://relay.example.test/v1/images/generations',
])
expect(result.operation).toBe('generate')
expect(result.inputImageCount).toBe(0)
})
test('uses the configured image model for a model-supplied default placeholder', async () => {
outputDir = await mkdtemp(join(tmpdir(), 'imagegen-output-'))
let requestBody: Record<string, unknown> | undefined
const fetchImpl = async (_input: string | URL | Request, init?: RequestInit) => {
requestBody = JSON.parse(String(init?.body))
return Response.json({
data: [{ b64_json: PNG_BYTES.toString('base64') }],
})
}
const result = await generateImages({
prompt: 'A paper-cut fox poster',
count: 1,
model: 'default',
}, customConfig, { fetchImpl, outputDir })
expect(requestBody).toMatchObject({ model: 'relay-image-model' })
expect(result.model).toBe('relay-image-model')
})
test('uses the configured ChatGPT OAuth image model for a default placeholder', async () => {
test('uses the configured ChatGPT OAuth image model', async () => {
outputDir = await mkdtemp(join(tmpdir(), 'imagegen-openai-oauth-'))
const tokenPath = join(outputDir, 'openai-oauth.json')
const previousTokenPath = process.env[OPENAI_CODEX_OAUTH_FILE_ENV_KEY]
@@ -318,7 +275,6 @@ describe('ImageGen backend', () => {
const result = await generateImages({
prompt: 'A paper-cut fox poster',
count: 1,
model: 'default',
}, {
kind: 'openai_oauth',
providerId: 'openai-official',
+2 -10
View File
@@ -30,7 +30,6 @@ export type ImageGenerationInput = {
path: string
role: 'edit_target' | 'reference' | 'style_reference' | 'composite_source'
}>
model?: string
aspect_ratio?: string
resolution?: '1k' | '2k'
size?: 'auto' | '1024x1024' | '1024x1536' | '1536x1024'
@@ -94,11 +93,7 @@ export async function generateImages(
options: GenerateOptions = {},
): Promise<ImageGenerationOutput> {
const startedAt = Date.now()
const requestedModel = input.model?.trim()
// Some tool-calling models use "default" to mean no model override.
const model = requestedModel && requestedModel.toLowerCase() !== 'default'
? requestedModel
: config.model
const model = config.model
const fetchImpl = options.fetchImpl ?? fetch
const { signal, cleanup } = createCombinedAbortSignal(options.signal, {
timeoutMs: IMAGE_REQUEST_TIMEOUT_MS,
@@ -611,10 +606,7 @@ async function prepareInputImages(
input: ImageGenerationInput,
overrideRootDirs?: string[],
): Promise<PreparedInputImage[]> {
// Some tool-calling models use the POSIX null device to mean "no image".
const requested = (input.input_images ?? []).filter(
({ path: inputPath }) => inputPath.trim() !== '/dev/null',
)
const requested = input.input_images ?? []
if (requested.length === 0) return []
if (requested.length > 3) {
throw new Error('Image editing supports at most 3 input images per call.')