fix(server): send images to OpenAI Chat models until the upstream rejects them

The OpenAI Chat proxy decided image support from host and model names:
opencode.ai and DeepSeek hosts replaced every image with a text-only
placeholder unless the model id contained "vision" or was allowlisted.
Each new vision model on OpenCode Go (Kimi K3, Space Bunny, DeepSeek Flash)
therefore told users the endpoint could not read images.

Offer images to every model instead. When the upstream refuses a request
that carried images with a recognizable image-input rejection, resend it
once without images and, only if that succeeds, remember the endpoint and
model as text-only for 30 minutes and record an info diagnostic. Failures
about a particular image (format, MIME type, size, animation, image count)
surface unchanged so one bad image cannot disable images for the model.

The rejection matcher moves to a shared module used by the CLI error mapper
and the proxy, and now also recognizes vLLM and OpenAI wording.
This commit is contained in:
程序员阿江(Relakkes)
2026-10-05 21:10:45 +08:00
parent 76af3f0e55
commit 4c9554cbcb
9 changed files with 593 additions and 93 deletions
+291 -20
View File
@@ -10,6 +10,8 @@ import { ProviderService } from '../services/providerService.js'
import { readActiveProviderManagedEnv } from '../services/providerRuntimeEnv.js'
import { handleProvidersApi } from '../api/providers.js'
import { handleProxyRequest } from '../proxy/handler.js'
import { resetOpenAIChatImageSupportForTests } from '../proxy/openaiChatImageSupport.js'
import { diagnosticsService } from '../services/diagnosticsService.js'
import {
clearTraceCaptureStateForTests,
drainTraceCaptureForTests,
@@ -32,11 +34,15 @@ async function setup() {
process.env.CLAUDE_CONFIG_DIR = tmpDir
process.env.HOME = tmpDir
clearTraceCaptureStateForTests()
resetOpenAIChatImageSupportForTests()
}
async function teardown() {
await drainTraceCaptureForTests()
clearTraceCaptureStateForTests()
// Diagnostics resolve their path when the queued write runs; flush them
// while CLAUDE_CONFIG_DIR still points at the temp dir.
await diagnosticsService.drainForMigration()
if (originalConfigDir !== undefined) {
process.env.CLAUDE_CONFIG_DIR = originalConfigDir
} else {
@@ -2304,33 +2310,294 @@ describe('ProviderService', () => {
test.each([
{
name: 'opencode non-vision model',
baseUrl: 'https://opencode.ai/zen',
name: 'DeepSeek schema error relayed by OpenCode Go',
baseUrl: 'https://opencode.ai/zen/go/v1',
model: 'deepseek-v4-flash',
status: 400,
error: 'Error from provider (DeepSeek): Failed to deserialize the JSON body into the target type: messages[0]: unknown variant `image_url`, expected `text` at line 1 column 120',
},
{
name: 'classic DeepSeek text model',
baseUrl: 'https://api.deepseek.com',
model: 'deepseek-v4-flash',
model: 'deepseek-v4-pro',
status: 400,
error: 'This model does not support image input',
},
])('uses text-only Computer Use content for $name', async ({ baseUrl, model }) => {
const body = await captureOpenAIChatRequest({
baseUrl,
model,
content: [{
type: 'image',
source: { type: 'base64', media_type: 'image/jpeg', data: 'private-screenshot-data' },
}],
})
{
name: 'generic gateway validation error',
baseUrl: 'https://gateway.example.test/v1',
model: 'text-model',
status: 422,
error: "messages.0.content.1.type: Input should be 'text'; received 'image_url'",
},
])('retries without images once $name rejects them, then stays text-only for that model', async ({ baseUrl, model, status, error }) => {
const originalFetch = globalThis.fetch
const calls: Array<Record<string, unknown>> = []
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
const body = JSON.parse(String(init?.body)) as Record<string, unknown>
calls.push(body)
if (JSON.stringify(body).includes('image_url')) {
return Response.json({ error: { type: 'invalid_request_error', message: error } }, { status })
}
return Response.json({
id: 'chatcmpl-text-only',
object: 'chat.completion',
created: 0,
model: body.model,
choices: [{ index: 0, message: { role: 'assistant', content: 'described from text' }, finish_reason: 'stop' }],
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
})
}) as typeof fetch
const messages = body.messages as Array<Record<string, unknown>>
expect(messages[0]).toEqual({
role: 'tool',
tool_call_id: 'computer_1',
content: '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n',
})
expect(JSON.stringify(body)).not.toContain('private-screenshot-data')
expect(JSON.stringify(body)).not.toContain('image_url')
try {
const otherModel = `${model}-sibling`
const svc = new ProviderService()
const provider = await svc.addProvider(sampleInput({
apiFormat: 'openai_chat',
baseUrl,
models: { main: model, haiku: otherModel, sonnet: model, opus: model },
}))
await svc.activateProvider(provider.id)
const send = (requestModel: string) => {
const req = new Request('http://localhost:3456/proxy/v1/messages', {
method: 'POST',
headers: { 'Content-Type': 'application/json', 'X-Claude-Code-Session-Id': `image-retry-${requestModel}` },
body: JSON.stringify({
model: requestModel,
max_tokens: 64,
messages: [{
role: 'user',
content: [
{ type: 'text', text: 'Describe this picture.' },
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'rejected-picture' } },
],
}],
}),
})
return handleProxyRequest(req, new URL(req.url))
}
const first = await send(model)
expect(first.status).toBe(200)
expect(await first.json()).toMatchObject({ content: [{ type: 'text', text: 'described from text' }] })
expect(calls).toHaveLength(2)
expect(JSON.stringify(calls[0])).toContain('data:image/png;base64,rejected-picture')
expect(calls[1].messages).toEqual([{
role: 'user',
content: 'Describe this picture.\n[Image omitted: the upstream rejected image input for this model.]\n',
}])
const trace = await waitForCompletedProxyTrace(`image-retry-${model}`)
expect(trace.calls.map(call => call.response?.status).sort()).toEqual([200, status].sort())
await diagnosticsService.drainForMigration()
const [learned] = (await diagnosticsService.readRecentEvents(20))
.filter(event => event.type === 'openai_chat_image_input_disabled')
expect(learned).toMatchObject({ severity: 'info', details: { model, httpStatus: status } })
expect(JSON.stringify(learned)).not.toContain('sk-test-key-123')
// The rejection is remembered: later turns skip the failing attempt.
const second = await send(model)
expect(second.status).toBe(200)
expect(calls).toHaveLength(3)
expect(JSON.stringify(calls[2])).not.toContain('image_url')
expect(JSON.stringify(calls[2])).not.toContain('rejected-picture')
// The learned limit belongs to that model, not to the whole endpoint.
await send(otherModel)
expect(calls[3].model).toBe(otherModel)
expect(JSON.stringify(calls[3])).toContain('data:image/png;base64,rejected-picture')
} finally {
globalThis.fetch = originalFetch
}
})
test.each([
{ name: 'an oversized image', status: 400, error: 'image exceeds maximum allowed size', withImage: true },
{
name: 'an unsupported image format',
status: 400,
error: "You uploaded an unsupported image. Please make sure your image has of one the following formats: ['png', 'jpeg', 'gif', 'webp'].",
withImage: true,
},
{ name: 'an unsupported image MIME type', status: 400, error: "Invalid image_url: unsupported MIME type 'image/heic'", withImage: true },
{ name: 'a server failure', status: 500, error: 'image input is not supported right now', withImage: true },
{ name: 'a request without images', status: 400, error: 'This model does not support image input', withImage: false },
])('surfaces $name without dropping images from later requests', async ({ status, error, withImage }) => {
const originalFetch = globalThis.fetch
const calls: Array<Record<string, unknown>> = []
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
calls.push(JSON.parse(String(init?.body)) as Record<string, unknown>)
return Response.json({ error: { type: 'invalid_request_error', message: error } }, { status })
}) as typeof fetch
try {
const svc = new ProviderService()
const provider = await svc.addProvider(sampleInput({
apiFormat: 'openai_chat',
baseUrl: 'https://opencode.ai/zen/go/v1',
models: { main: 'deepseek-flash', haiku: 'deepseek-flash', sonnet: 'deepseek-flash', opus: 'deepseek-flash' },
}))
await svc.activateProvider(provider.id)
const send = (content: unknown) => {
const req = new Request('http://localhost:3456/proxy/v1/messages', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ model: 'deepseek-flash', max_tokens: 64, messages: [{ role: 'user', content }] }),
})
return handleProxyRequest(req, new URL(req.url))
}
const image = { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'kept-picture' } }
const res = await send(withImage ? [{ type: 'text', text: 'Look.' }, image] : 'Look.')
expect(res.status).toBe(status)
expect(JSON.stringify(await res.json())).toContain(error)
expect(calls).toHaveLength(1)
// The next turn still offers its image to the model first.
await send([{ type: 'text', text: 'Look again.' }, image])
expect(JSON.stringify(calls[1])).toContain('data:image/png;base64,kept-picture')
} finally {
globalThis.fetch = originalFetch
}
})
test.each([
'https://api.deepseek.com',
'https://opencode.ai/zen/v1',
'https://opencode.ai/zen/go/v1',
'https://gateway.example.test/v1',
].flatMap(baseUrl => ['deepseek-v4-pro', 'deepseek-flash', 'glm-5.3'].map(model => ({ baseUrl, model }))))(
'offers images to $model at $baseUrl before any rejection',
async ({ baseUrl, model }) => {
// Guard: image support must never be predicted from host or model names.
const body = await captureOpenAIChatRequest({
baseUrl,
model,
contentSource: 'user',
content: [
{ type: 'text', text: 'Look.' },
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'guard-picture' } },
],
})
expect(JSON.stringify(body)).toContain('data:image/png;base64,guard-picture')
expect(JSON.stringify(body)).not.toContain('Image omitted:')
},
)
test('keeps offering images when the text-only resend fails for another reason', async () => {
const originalFetch = globalThis.fetch
const calls: Array<Record<string, unknown>> = []
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
const body = JSON.parse(String(init?.body)) as Record<string, unknown>
calls.push(body)
return JSON.stringify(body).includes('image_url')
? Response.json({ error: { message: 'This model does not support image input' } }, { status: 400 })
: Response.json({ error: { message: 'Rate limit reached' } }, { status: 429 })
}) as typeof fetch
try {
const svc = new ProviderService()
const provider = await svc.addProvider(sampleInput({
apiFormat: 'openai_chat',
baseUrl: 'https://opencode.ai/zen/go/v1',
models: { main: 'deepseek-flash', haiku: 'deepseek-flash', sonnet: 'deepseek-flash', opus: 'deepseek-flash' },
}))
await svc.activateProvider(provider.id)
const send = () => {
const req = new Request('http://localhost:3456/proxy/v1/messages', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model: 'deepseek-flash',
max_tokens: 64,
messages: [{
role: 'user',
content: [
{ type: 'text', text: 'Look.' },
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'retry-picture' } },
],
}],
}),
})
return handleProxyRequest(req, new URL(req.url))
}
const first = await send()
expect(first.status).toBe(429)
expect(calls).toHaveLength(2)
await send()
expect(JSON.stringify(calls[2])).toContain('data:image/png;base64,retry-picture')
} finally {
globalThis.fetch = originalFetch
}
})
test('resends a streaming request without images after an image rejection', async () => {
const originalFetch = globalThis.fetch
const encoder = new TextEncoder()
const calls: Array<Record<string, unknown>> = []
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
const body = JSON.parse(String(init?.body)) as Record<string, unknown>
calls.push(body)
if (JSON.stringify(body).includes('image_url')) {
return Response.json({
error: { message: 'Error from provider (DeepSeek): unknown variant `image_url`, expected `text`' },
}, { status: 400 })
}
return new Response(new ReadableStream<Uint8Array>({
start(controller) {
controller.enqueue(encoder.encode([
'data: {"id":"chatcmpl-stream-retry","object":"chat.completion.chunk","model":"deepseek-flash","choices":[{"index":0,"delta":{"role":"assistant","content":"text answer"},"finish_reason":null}]}',
'',
'data: {"id":"chatcmpl-stream-retry","object":"chat.completion.chunk","model":"deepseek-flash","choices":[{"index":0,"delta":{},"finish_reason":"stop"}],"usage":{"prompt_tokens":5,"completion_tokens":2,"total_tokens":7}}',
'',
'data: [DONE]',
'',
].join('\n')))
controller.close()
},
}), { status: 200, headers: { 'Content-Type': 'text/event-stream' } })
}) as typeof fetch
try {
const svc = new ProviderService()
const provider = await svc.addProvider(sampleInput({
apiFormat: 'openai_chat',
baseUrl: 'https://opencode.ai/zen/go/v1',
models: { main: 'deepseek-flash', haiku: 'deepseek-flash', sonnet: 'deepseek-flash', opus: 'deepseek-flash' },
}))
await svc.activateProvider(provider.id)
const req = new Request('http://localhost:3456/proxy/v1/messages', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model: 'deepseek-flash[1m]',
max_tokens: 64,
stream: true,
messages: [{
role: 'user',
content: [
{ type: 'text', text: 'Look.' },
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'stream-picture' } },
],
}],
}),
})
const res = await handleProxyRequest(req, new URL(req.url))
expect(res.status).toBe(200)
expect(res.headers.get('Content-Type')).toBe('text/event-stream')
const text = await res.text()
expect(text).toContain('text answer')
expect(text).toContain('message_stop')
expect(calls).toHaveLength(2)
expect(calls[1].stream).toBe(true)
expect(JSON.stringify(calls[1])).not.toContain('stream-picture')
} finally {
globalThis.fetch = originalFetch
}
})
test.each([
@@ -2365,6 +2632,10 @@ describe('ProviderService', () => {
{ model: 'space-bunny-free', contentSource: 'tool' },
{ model: 'space-bunny-free[1m]', contentSource: 'user' },
{ model: 'space-bunny-free[1m]', contentSource: 'tool' },
{ model: 'deepseek-flash', contentSource: 'user' },
{ model: 'deepseek-flash', contentSource: 'tool' },
{ model: 'deepseek-flash[1m]', contentSource: 'user' },
{ model: 'some-future-model', contentSource: 'tool' },
] as const)('forwards OpenCode Go images for $model from $contentSource content', async ({ model, contentSource }) => {
const body = await captureOpenAIChatRequest({
baseUrl: 'https://opencode.ai/zen/go/v1',
+3 -3
View File
@@ -865,7 +865,7 @@ describe('anthropicToOpenaiChat', () => {
}
const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' })
expect(result.messages[0].content).toBe(
'What is in this screenshot?\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n',
'What is in this screenshot?\n[Image omitted: the upstream rejected image input for this model.]\n',
)
expect(JSON.stringify(result)).not.toContain('image_url')
expect(JSON.stringify(result)).not.toContain('abc123')
@@ -889,7 +889,7 @@ describe('anthropicToOpenaiChat', () => {
}
const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' })
expect(result.messages).toEqual([
{ role: 'tool', tool_call_id: 'tc_img', content: '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n' },
{ role: 'tool', tool_call_id: 'tc_img', content: '\n[Image omitted: the upstream rejected image input for this model.]\n' },
])
expect(JSON.stringify(result)).not.toContain('image_url')
expect(JSON.stringify(result)).not.toContain('abc123')
@@ -946,7 +946,7 @@ describe('anthropicToOpenaiChat', () => {
expect(result.messages).toEqual([
{
role: 'user',
content: '[Document: cited]\nbefore\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\nafter',
content: '[Document: cited]\nbefore\n[Image omitted: the upstream rejected image input for this model.]\nafter',
},
])
expect(JSON.stringify(result)).not.toContain('image_url')
+45 -26
View File
@@ -15,9 +15,15 @@ import { normalizeAnthropicBaseUrl } from '../../services/api/anthropicBaseUrl.j
import { createGunzip, createInflate } from 'node:zlib'
import { ProviderService } from '../services/providerService.js'
import { diagnosticsService } from '../services/diagnosticsService.js'
import type { ProviderAuthStrategy } from '../types/provider.js'
import { resolvePromptCacheKey } from './promptCacheKey.js'
import { anthropicToOpenaiChat } from './transform/anthropicToOpenaiChat.js'
import {
getOpenAIChatImageContentMode,
isOpenAIChatImageRejection,
rememberOpenAIChatTextOnlyModel,
} from './openaiChatImageSupport.js'
import { anthropicToOpenaiChat, type OpenAIChatImageContentMode } from './transform/anthropicToOpenaiChat.js'
import { anthropicToOpenaiResponses } from './transform/anthropicToOpenaiResponses.js'
import { RequestCompatibilityError, resolveRequestCompatibility, type RequestCompatibilityOptions } from './transform/requestCompatibility.js'
import { ProtocolTraceObserver, observeProtocolStream, type ProtocolTraceTransport } from './protocolTrace.js'
@@ -698,20 +704,22 @@ async function handleOpenaiChat(
traceContext: ProxyTraceContext | null,
requestOptions: RequestCompatibilityOptions = {},
upstreamHeaders: Record<string, string> = {},
imageContentModeOverride?: OpenAIChatImageContentMode,
): Promise<Response> {
const knownDeepSeekHost = shouldUseDeepSeekReasoningCompat(baseUrl)
const reasoningProfile = resolveModelReasoningProfile(body.model, 'openai_chat')
const url = buildOpenaiEndpoint(baseUrl, 'chat/completions')
const imageContentMode = imageContentModeOverride ?? getOpenAIChatImageContentMode(url, body.model)
const transformed = anthropicToOpenaiChat(body, {
...requestOptions,
roundTripReasoningContent: knownDeepSeekHost || reasoningProfile?.family === 'deepseek-v4',
passThinkingToggle: knownDeepSeekHost,
imageContentMode: shouldUseTextOnlyOpenAIChatContent(baseUrl, body.model) ? 'text_only' : 'vision',
imageContentMode,
})
if (traceContext) {
traceContext.protocolTrace = new ProtocolTraceObserver('openai_chat', transformed,
resolveRequestCompatibility(body, { ...requestOptions, protocol: 'openai_chat' }).outputBudget)
}
const url = buildOpenaiEndpoint(baseUrl, 'chat/completions')
// Preset-declared headers first: `Authorization` is applied last so a preset can
// never shadow the credential, and the resolver already drops framing headers.
const upstreamRequestHeaders: Record<string, string> = {}
@@ -760,6 +768,40 @@ async function handleOpenaiChat(
if (!upstream.ok) {
const errText = await upstream.text().catch(() => '')
if (imageContentMode === 'vision' && isOpenAIChatImageRejection(upstream.status, errText, transformed)) {
// The upstream refused the request before generating anything, so the
// client has seen no output and resending is safe. The model is only
// remembered once the same request succeeds without images, so an
// unrelated failure of the resend cannot disable images for good.
if (traceContext) {
recordProxyTraceInBackground({
context: traceContext,
callId: traceCallId,
model: body.model,
upstreamUrl: url,
upstreamRequest: transformed,
requestHeaders: upstreamRequestHeaders,
startedAt,
startedAtMs,
responseStatus: upstream.status,
upstreamResponseBody: errText,
responseHeaders: upstream.headers,
})
}
const resent = await handleOpenaiChat(
body, baseUrl, apiKey, isStream, networkSettings, traceContext, requestOptions, upstreamHeaders, 'text_only',
)
if (resent.ok) {
rememberOpenAIChatTextOnlyModel(url, body.model)
void diagnosticsService.recordEvent({
type: 'openai_chat_image_input_disabled',
severity: 'info',
summary: `Upstream rejected image input for ${body.model}; images are omitted for this model for 30 minutes`,
details: { model: body.model, upstreamUrl: url, httpStatus: upstream.status, upstreamError: errText.slice(0, 500) },
})
}
return resent
}
let policyError = null
try {
policyError = getOpenAIPolicyError(JSON.parse(errText))
@@ -880,29 +922,6 @@ function shouldUseDeepSeekReasoningCompat(baseUrl: string): boolean {
)
}
function shouldUseTextOnlyOpenAIChatContent(baseUrl: string, model: string): boolean {
// Keep classic DeepSeek text models compatible without dropping images for
// explicitly vision-capable models served by the same Chat endpoint.
if (/(^|[./-])deepseek([./-]|$)/i.test(baseUrl)) {
return !hasExplicitVisionModelMarker(model)
}
// OpenCode Go's Kimi K3 and Space Bunny Free accept image_url despite lacking
// "vision" in their model ids. Keep other unverified gateway models on the
// text-only path. Context-window suffixes have already been stripped.
if (/(^|[./-])opencode\.ai([:/]|$)/i.test(baseUrl)) {
return !hasExplicitVisionModelMarker(model) && !['kimi-k3', 'space-bunny-free'].includes(model.toLowerCase())
}
// Preserve the existing behavior for generic compatible providers whose
// capabilities are not controlled by either compatibility policy above.
return false
}
function hasExplicitVisionModelMarker(model: string): boolean {
return /(^|[/:._-])vision([/:._-]|$)/i.test(model)
}
async function handleOpenaiResponses(
body: AnthropicRequest,
baseUrl: string,
@@ -0,0 +1,86 @@
import { afterEach, describe, expect, test } from 'bun:test'
import {
getOpenAIChatImageContentMode,
isOpenAIChatImageRejection,
rememberOpenAIChatTextOnlyModel,
resetOpenAIChatImageSupportForTests,
} from './openaiChatImageSupport.js'
import type { OpenAIChatRequest } from './transform/types.js'
const withImage: OpenAIChatRequest = {
model: 'deepseek-flash',
messages: [{
role: 'user',
content: [
{ type: 'text', text: 'Look.' },
{ type: 'image_url', image_url: { url: 'data:image/png;base64,AAAA' } },
],
}],
}
const withoutImage: OpenAIChatRequest = { model: 'deepseek-flash', messages: [{ role: 'user', content: 'Look.' }] }
afterEach(resetOpenAIChatImageSupportForTests)
describe('isOpenAIChatImageRejection', () => {
test('accepts an explicit refusal of image input', () => {
const refusals = [
'Error from provider (DeepSeek): Failed to deserialize the JSON body into the target type: messages[0]: unknown variant `image_url`, expected `text` at line 1 column 120',
'Error from provider (Moonshot): This model does not support image input',
"messages.0.content.1.type: Input should be 'text'; received 'image_url'",
'unsupported modality: image input is not available',
'Qwen/Qwen3-8B is not a multimodal model',
'Invalid content type. image_url is only supported by certain models.',
]
for (const refusal of refusals) {
expect(isOpenAIChatImageRejection(400, refusal, withImage)).toBe(true)
}
expect(isOpenAIChatImageRejection(422, refusals[2]!, withImage)).toBe(true)
})
test('rejects failures about one particular image', () => {
const imageProblems = [
"You uploaded an unsupported image. Please make sure your image has of one the following formats: ['png', 'jpeg', 'gif', 'webp'].",
"Invalid image_url: unsupported MIME type 'image/heic'",
'unsupported image format: image/bmp',
'Image format not supported',
'Image is not supported: exceeds the 20 MB limit',
'image_url is not supported: failed to download the image',
'Unsupported image: could not decode the image data',
'Multiple images are not supported by this model',
'Animated images are not supported',
'Image type gif is not supported',
'Image input is not supported in your region',
]
for (const problem of imageProblems) {
expect(isOpenAIChatImageRejection(400, problem, withImage)).toBe(false)
}
})
test('rejects server failures and requests without images', () => {
const refusal = 'This model does not support image input'
expect(isOpenAIChatImageRejection(500, refusal, withImage)).toBe(false)
expect(isOpenAIChatImageRejection(429, refusal, withImage)).toBe(false)
expect(isOpenAIChatImageRejection(400, refusal, withoutImage)).toBe(false)
})
})
describe('learned text-only models', () => {
test('are scoped to one endpoint and model, ignoring case and trailing slashes', () => {
rememberOpenAIChatTextOnlyModel('https://opencode.ai/zen/go/v1/chat/completions/', 'DeepSeek-V4-Pro')
expect(getOpenAIChatImageContentMode('https://opencode.ai/zen/go/v1/chat/completions', 'deepseek-v4-pro')).toBe('text_only')
expect(getOpenAIChatImageContentMode('https://opencode.ai/zen/go/v1/chat/completions', 'deepseek-flash')).toBe('vision')
expect(getOpenAIChatImageContentMode('https://api.deepseek.com/chat/completions', 'deepseek-v4-pro')).toBe('vision')
})
test('expire after 30 minutes so the model is probed with images again', () => {
const endpoint = 'https://opencode.ai/zen/go/v1/chat/completions'
const learnedAt = 1_000_000
rememberOpenAIChatTextOnlyModel(endpoint, 'deepseek-v4-pro', learnedAt)
expect(getOpenAIChatImageContentMode(endpoint, 'deepseek-v4-pro', learnedAt + 29 * 60_000)).toBe('text_only')
expect(getOpenAIChatImageContentMode(endpoint, 'deepseek-v4-pro', learnedAt + 31 * 60_000)).toBe('vision')
})
})
@@ -0,0 +1,80 @@
import { isUnsupportedImageInputErrorMessage } from '../../services/api/unsupportedImageInput.js'
import type { OpenAIChatImageContentMode } from './transform/anthropicToOpenaiChat.js'
import type { OpenAIChatRequest } from './transform/types.js'
// Image support is learned from the upstream, not predicted from host or model
// names: gateways such as OpenCode Go add vision models faster than any static
// list can follow. Every model starts on the vision path; a model is moved to
// text-only only after its endpoint explicitly rejects image input and the
// same request succeeds without images. The knowledge expires, so a model that
// gains vision support (or a rare misreading) recovers without a restart.
const TEXT_ONLY_TTL_MS = 30 * 60 * 1000
const MAX_TEXT_ONLY_MODELS = 256
const textOnlyModelsUntil = new Map<string, number>()
// Failures about particular images (format, MIME type, size, animation, image
// count, broken data, unreachable URL) also mention images, but the model
// still accepts other images. Learning from them would strip every later image
// for this model, so they surface unchanged; a missed refusal only costs one
// visible error.
const IMAGE_CONTENT_PROBLEM = new RegExp([
String.raw`\bmime\b`, String.raw`\bformats?\b`, String.raw`\bgif\b`, 'animated', 'image type',
'multiple images', 'too many images', 'number of images',
'too large', 'exceed', String.raw`\bsize\b`, 'dimension', 'resolution', 'pixel',
'corrupt', 'decod', 'base64', 'download', 'fetch', 'could not process', 'unable to process', 'region',
].join('|'), 'i')
function modelKey(endpointUrl: string, model: string): string {
return `${endpointUrl.replace(/\/+$/, '').toLowerCase()}\n${model.toLowerCase()}`
}
export function getOpenAIChatImageContentMode(
endpointUrl: string,
model: string,
now = Date.now(),
): OpenAIChatImageContentMode {
const key = modelKey(endpointUrl, model)
const until = textOnlyModelsUntil.get(key)
if (until === undefined) return 'vision'
if (until > now) return 'text_only'
textOnlyModelsUntil.delete(key)
return 'vision'
}
export function rememberOpenAIChatTextOnlyModel(endpointUrl: string, model: string, now = Date.now()): void {
const key = modelKey(endpointUrl, model)
textOnlyModelsUntil.delete(key)
if (textOnlyModelsUntil.size >= MAX_TEXT_ONLY_MODELS) {
const oldest = textOnlyModelsUntil.keys().next().value
if (oldest !== undefined) textOnlyModelsUntil.delete(oldest)
}
textOnlyModelsUntil.set(key, now + TEXT_ONLY_TTL_MS)
}
/**
* Whether a failed upstream response is the endpoint refusing image input for
* this model. Only an explicit refusal of a request that actually carried
* images qualifies: other failures (a bad image, context overflow, 5xx) must
* surface unchanged instead of silently dropping the user's images.
*/
export function isOpenAIChatImageRejection(
status: number,
errorText: string,
request: OpenAIChatRequest,
): boolean {
return (
(status === 400 || status === 422) &&
requestCarriesImages(request) &&
isUnsupportedImageInputErrorMessage(errorText) &&
!IMAGE_CONTENT_PROBLEM.test(errorText)
)
}
function requestCarriesImages(request: OpenAIChatRequest): boolean {
return request.messages.some(message =>
Array.isArray(message.content) && message.content.some(part => part.type === 'image_url'))
}
export function resetOpenAIChatImageSupportForTests(): void {
textOnlyModelsUntil.clear()
}
@@ -18,7 +18,7 @@ import { stripLeadingBillingHeader } from './billingHeader.js'
import { normalizeOpenAIReasoningEffort } from './effort.js'
import { resolveRequestCompatibility, type RequestCompatibilityOptions } from './requestCompatibility.js'
type OpenAIChatImageContentMode = 'vision' | 'text_only'
export type OpenAIChatImageContentMode = 'vision' | 'text_only'
// Synthetic text parts (degraded documents, search results) carry an internal
// marker so serializers can preserve their boundaries. The marker is removed
@@ -35,7 +35,7 @@ export type OpenAIChatTransformOptions = RequestCompatibilityOptions & {
// Synthetic degradation text carries its own separators: the parts are
// joined without a separator, so a notice must not glue itself to the
// surrounding user text.
const OMITTED_IMAGE_TEXT = '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n'
const OMITTED_IMAGE_TEXT = '\n[Image omitted: the upstream rejected image input for this model.]\n'
const FILE_IMAGE_OMITTED_TEXT = '\n[Image omitted: file-based image source is not supported by this endpoint.]\n'
const MEDIA_RESULT_ATTACHED_TEXT = 'Media result attached after this tool result.'
const DOCUMENT_TEXT_INLINE_LIMIT = 2000
+3 -42
View File
@@ -56,6 +56,9 @@ import {
StreamEndedEarlyError,
} from './streamFallback.js'
import { StreamWatchdogTimeoutError } from './streamWatchdog.js'
import { isUnsupportedImageInputErrorMessage } from './unsupportedImageInput.js'
export { isUnsupportedImageInputErrorMessage }
// Presentation only: classifiers, retries and diagnostic metadata keep the
// original SDK error. Decode envelopes structurally so escaped quotes/newlines
@@ -596,48 +599,6 @@ export function extractUnknownErrorFormat(value: unknown): string | undefined {
return undefined
}
export function isUnsupportedImageInputErrorMessage(message: string): boolean {
const raw = message.toLowerCase()
if (!raw.includes('image')) return false
if (isOpenAIImageUrlTextOnlySchemaError(raw)) {
return true
}
return (
raw.includes('not support') ||
raw.includes('not supported') ||
raw.includes('unsupported') ||
raw.includes('vision') ||
raw.includes('multimodal') ||
raw.includes('multi-modal') ||
raw.includes('modality')
)
}
function isOpenAIImageUrlTextOnlySchemaError(raw: string): boolean {
if (!raw.includes('image_url')) return false
if (
raw.includes('not allowed') ||
raw.includes('not permitted') ||
raw.includes('disallowed') ||
raw.includes('forbidden')
) {
return true
}
if (!raw.includes('text')) return false
return (
raw.includes('expected') ||
raw.includes('input should be') ||
raw.includes('not one of') ||
raw.includes('permitted') ||
raw.includes('received') ||
raw.includes('unknown variant') ||
raw.includes('invalid value') ||
raw.includes('invalid type') ||
raw.includes('valid enumeration') ||
raw.includes('only text')
)
}
// Deep-walk a request payload looking for image blocks. Images can sit at the
// top level of a user message or nested inside a tool_result's content, and
// the payloads here are { type, message: { content } } wrappers, so walk all
@@ -0,0 +1,33 @@
import { describe, expect, test } from 'bun:test'
import { isUnsupportedImageInputErrorMessage } from './unsupportedImageInput.js'
describe('isUnsupportedImageInputErrorMessage', () => {
test('recognizes rejections relayed through a gateway prefix', () => {
// OpenCode Zen/Go wraps the upstream message as "Error from provider (<name>): <message>".
const relayed = [
'Error from provider (DeepSeek): Failed to deserialize the JSON body into the target type: messages[2]: unknown variant `image_url`, expected `text` at line 1 column 512',
'Error from provider (Moonshot): This model does not support image input',
'{"error":{"message":"Error from provider: Invalid value for \'messages[0].content[1].type\': \'image_url\' is not one of [\'text\']"}}',
// vLLM names the modality instead of images.
'Qwen/Qwen3-8B is not a multimodal model',
'Invalid content type. image_url is only supported by certain models.',
]
for (const message of relayed) {
expect(isUnsupportedImageInputErrorMessage(message)).toBe(true)
}
})
test('ignores image failures that are not about the model refusing images', () => {
const unrelated = [
'image exceeds maximum',
'Error from provider (DeepSeek): Image is too large: 34 MB exceeds the 20 MB limit',
'Error from provider (DeepSeek): This model maximum context length is 131072 tokens',
'unsupported content block type: only text is allowed for this model',
]
for (const message of unrelated) {
expect(isUnsupportedImageInputErrorMessage(message)).toBe(false)
}
})
})
+50
View File
@@ -0,0 +1,50 @@
/**
* Recognizes provider errors that reject image input for a text-only model.
*
* Kept free of CLI state so both the CLI error mapper and the desktop proxy
* (which retries OpenAI Chat requests without images) share one definition.
*/
export function isUnsupportedImageInputErrorMessage(message: string): boolean {
const raw = message.toLowerCase()
// vLLM: "<model> is not a multimodal model" names the modality, not images.
if (/\bnot an? (multimodal|multi-modal|vision) model\b/.test(raw)) return true
if (!raw.includes('image')) return false
if (isOpenAIImageUrlTextOnlySchemaError(raw)) {
return true
}
return (
raw.includes('not support') ||
raw.includes('not supported') ||
raw.includes('unsupported') ||
raw.includes('vision') ||
raw.includes('multimodal') ||
raw.includes('multi-modal') ||
raw.includes('modality')
)
}
function isOpenAIImageUrlTextOnlySchemaError(raw: string): boolean {
if (!raw.includes('image_url')) return false
if (
raw.includes('not allowed') ||
raw.includes('not permitted') ||
raw.includes('disallowed') ||
raw.includes('forbidden') ||
raw.includes('only supported by certain models')
) {
return true
}
if (!raw.includes('text')) return false
return (
raw.includes('expected') ||
raw.includes('input should be') ||
raw.includes('not one of') ||
raw.includes('permitted') ||
raw.includes('received') ||
raw.includes('unknown variant') ||
raw.includes('invalid value') ||
raw.includes('invalid type') ||
raw.includes('valid enumeration') ||
raw.includes('only text')
)
}