mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 03:43:11 +08:00
fix(server): send images to OpenAI Chat models until the upstream rejects them
The OpenAI Chat proxy decided image support from host and model names: opencode.ai and DeepSeek hosts replaced every image with a text-only placeholder unless the model id contained "vision" or was allowlisted. Each new vision model on OpenCode Go (Kimi K3, Space Bunny, DeepSeek Flash) therefore told users the endpoint could not read images. Offer images to every model instead. When the upstream refuses a request that carried images with a recognizable image-input rejection, resend it once without images and, only if that succeeds, remember the endpoint and model as text-only for 30 minutes and record an info diagnostic. Failures about a particular image (format, MIME type, size, animation, image count) surface unchanged so one bad image cannot disable images for the model. The rejection matcher moves to a shared module used by the CLI error mapper and the proxy, and now also recognizes vLLM and OpenAI wording.
This commit is contained in:
@@ -10,6 +10,8 @@ import { ProviderService } from '../services/providerService.js'
|
||||
import { readActiveProviderManagedEnv } from '../services/providerRuntimeEnv.js'
|
||||
import { handleProvidersApi } from '../api/providers.js'
|
||||
import { handleProxyRequest } from '../proxy/handler.js'
|
||||
import { resetOpenAIChatImageSupportForTests } from '../proxy/openaiChatImageSupport.js'
|
||||
import { diagnosticsService } from '../services/diagnosticsService.js'
|
||||
import {
|
||||
clearTraceCaptureStateForTests,
|
||||
drainTraceCaptureForTests,
|
||||
@@ -32,11 +34,15 @@ async function setup() {
|
||||
process.env.CLAUDE_CONFIG_DIR = tmpDir
|
||||
process.env.HOME = tmpDir
|
||||
clearTraceCaptureStateForTests()
|
||||
resetOpenAIChatImageSupportForTests()
|
||||
}
|
||||
|
||||
async function teardown() {
|
||||
await drainTraceCaptureForTests()
|
||||
clearTraceCaptureStateForTests()
|
||||
// Diagnostics resolve their path when the queued write runs; flush them
|
||||
// while CLAUDE_CONFIG_DIR still points at the temp dir.
|
||||
await diagnosticsService.drainForMigration()
|
||||
if (originalConfigDir !== undefined) {
|
||||
process.env.CLAUDE_CONFIG_DIR = originalConfigDir
|
||||
} else {
|
||||
@@ -2304,33 +2310,294 @@ describe('ProviderService', () => {
|
||||
|
||||
test.each([
|
||||
{
|
||||
name: 'opencode non-vision model',
|
||||
baseUrl: 'https://opencode.ai/zen',
|
||||
name: 'DeepSeek schema error relayed by OpenCode Go',
|
||||
baseUrl: 'https://opencode.ai/zen/go/v1',
|
||||
model: 'deepseek-v4-flash',
|
||||
status: 400,
|
||||
error: 'Error from provider (DeepSeek): Failed to deserialize the JSON body into the target type: messages[0]: unknown variant `image_url`, expected `text` at line 1 column 120',
|
||||
},
|
||||
{
|
||||
name: 'classic DeepSeek text model',
|
||||
baseUrl: 'https://api.deepseek.com',
|
||||
model: 'deepseek-v4-flash',
|
||||
model: 'deepseek-v4-pro',
|
||||
status: 400,
|
||||
error: 'This model does not support image input',
|
||||
},
|
||||
])('uses text-only Computer Use content for $name', async ({ baseUrl, model }) => {
|
||||
const body = await captureOpenAIChatRequest({
|
||||
baseUrl,
|
||||
model,
|
||||
content: [{
|
||||
type: 'image',
|
||||
source: { type: 'base64', media_type: 'image/jpeg', data: 'private-screenshot-data' },
|
||||
}],
|
||||
})
|
||||
{
|
||||
name: 'generic gateway validation error',
|
||||
baseUrl: 'https://gateway.example.test/v1',
|
||||
model: 'text-model',
|
||||
status: 422,
|
||||
error: "messages.0.content.1.type: Input should be 'text'; received 'image_url'",
|
||||
},
|
||||
])('retries without images once $name rejects them, then stays text-only for that model', async ({ baseUrl, model, status, error }) => {
|
||||
const originalFetch = globalThis.fetch
|
||||
const calls: Array<Record<string, unknown>> = []
|
||||
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
|
||||
const body = JSON.parse(String(init?.body)) as Record<string, unknown>
|
||||
calls.push(body)
|
||||
if (JSON.stringify(body).includes('image_url')) {
|
||||
return Response.json({ error: { type: 'invalid_request_error', message: error } }, { status })
|
||||
}
|
||||
return Response.json({
|
||||
id: 'chatcmpl-text-only',
|
||||
object: 'chat.completion',
|
||||
created: 0,
|
||||
model: body.model,
|
||||
choices: [{ index: 0, message: { role: 'assistant', content: 'described from text' }, finish_reason: 'stop' }],
|
||||
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
|
||||
})
|
||||
}) as typeof fetch
|
||||
|
||||
const messages = body.messages as Array<Record<string, unknown>>
|
||||
expect(messages[0]).toEqual({
|
||||
role: 'tool',
|
||||
tool_call_id: 'computer_1',
|
||||
content: '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n',
|
||||
})
|
||||
expect(JSON.stringify(body)).not.toContain('private-screenshot-data')
|
||||
expect(JSON.stringify(body)).not.toContain('image_url')
|
||||
try {
|
||||
const otherModel = `${model}-sibling`
|
||||
const svc = new ProviderService()
|
||||
const provider = await svc.addProvider(sampleInput({
|
||||
apiFormat: 'openai_chat',
|
||||
baseUrl,
|
||||
models: { main: model, haiku: otherModel, sonnet: model, opus: model },
|
||||
}))
|
||||
await svc.activateProvider(provider.id)
|
||||
const send = (requestModel: string) => {
|
||||
const req = new Request('http://localhost:3456/proxy/v1/messages', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json', 'X-Claude-Code-Session-Id': `image-retry-${requestModel}` },
|
||||
body: JSON.stringify({
|
||||
model: requestModel,
|
||||
max_tokens: 64,
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: [
|
||||
{ type: 'text', text: 'Describe this picture.' },
|
||||
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'rejected-picture' } },
|
||||
],
|
||||
}],
|
||||
}),
|
||||
})
|
||||
return handleProxyRequest(req, new URL(req.url))
|
||||
}
|
||||
|
||||
const first = await send(model)
|
||||
expect(first.status).toBe(200)
|
||||
expect(await first.json()).toMatchObject({ content: [{ type: 'text', text: 'described from text' }] })
|
||||
expect(calls).toHaveLength(2)
|
||||
expect(JSON.stringify(calls[0])).toContain('data:image/png;base64,rejected-picture')
|
||||
expect(calls[1].messages).toEqual([{
|
||||
role: 'user',
|
||||
content: 'Describe this picture.\n[Image omitted: the upstream rejected image input for this model.]\n',
|
||||
}])
|
||||
|
||||
const trace = await waitForCompletedProxyTrace(`image-retry-${model}`)
|
||||
expect(trace.calls.map(call => call.response?.status).sort()).toEqual([200, status].sort())
|
||||
await diagnosticsService.drainForMigration()
|
||||
const [learned] = (await diagnosticsService.readRecentEvents(20))
|
||||
.filter(event => event.type === 'openai_chat_image_input_disabled')
|
||||
expect(learned).toMatchObject({ severity: 'info', details: { model, httpStatus: status } })
|
||||
expect(JSON.stringify(learned)).not.toContain('sk-test-key-123')
|
||||
|
||||
// The rejection is remembered: later turns skip the failing attempt.
|
||||
const second = await send(model)
|
||||
expect(second.status).toBe(200)
|
||||
expect(calls).toHaveLength(3)
|
||||
expect(JSON.stringify(calls[2])).not.toContain('image_url')
|
||||
expect(JSON.stringify(calls[2])).not.toContain('rejected-picture')
|
||||
|
||||
// The learned limit belongs to that model, not to the whole endpoint.
|
||||
await send(otherModel)
|
||||
expect(calls[3].model).toBe(otherModel)
|
||||
expect(JSON.stringify(calls[3])).toContain('data:image/png;base64,rejected-picture')
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch
|
||||
}
|
||||
})
|
||||
|
||||
test.each([
|
||||
{ name: 'an oversized image', status: 400, error: 'image exceeds maximum allowed size', withImage: true },
|
||||
{
|
||||
name: 'an unsupported image format',
|
||||
status: 400,
|
||||
error: "You uploaded an unsupported image. Please make sure your image has of one the following formats: ['png', 'jpeg', 'gif', 'webp'].",
|
||||
withImage: true,
|
||||
},
|
||||
{ name: 'an unsupported image MIME type', status: 400, error: "Invalid image_url: unsupported MIME type 'image/heic'", withImage: true },
|
||||
{ name: 'a server failure', status: 500, error: 'image input is not supported right now', withImage: true },
|
||||
{ name: 'a request without images', status: 400, error: 'This model does not support image input', withImage: false },
|
||||
])('surfaces $name without dropping images from later requests', async ({ status, error, withImage }) => {
|
||||
const originalFetch = globalThis.fetch
|
||||
const calls: Array<Record<string, unknown>> = []
|
||||
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
|
||||
calls.push(JSON.parse(String(init?.body)) as Record<string, unknown>)
|
||||
return Response.json({ error: { type: 'invalid_request_error', message: error } }, { status })
|
||||
}) as typeof fetch
|
||||
|
||||
try {
|
||||
const svc = new ProviderService()
|
||||
const provider = await svc.addProvider(sampleInput({
|
||||
apiFormat: 'openai_chat',
|
||||
baseUrl: 'https://opencode.ai/zen/go/v1',
|
||||
models: { main: 'deepseek-flash', haiku: 'deepseek-flash', sonnet: 'deepseek-flash', opus: 'deepseek-flash' },
|
||||
}))
|
||||
await svc.activateProvider(provider.id)
|
||||
const send = (content: unknown) => {
|
||||
const req = new Request('http://localhost:3456/proxy/v1/messages', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ model: 'deepseek-flash', max_tokens: 64, messages: [{ role: 'user', content }] }),
|
||||
})
|
||||
return handleProxyRequest(req, new URL(req.url))
|
||||
}
|
||||
const image = { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'kept-picture' } }
|
||||
|
||||
const res = await send(withImage ? [{ type: 'text', text: 'Look.' }, image] : 'Look.')
|
||||
expect(res.status).toBe(status)
|
||||
expect(JSON.stringify(await res.json())).toContain(error)
|
||||
expect(calls).toHaveLength(1)
|
||||
|
||||
// The next turn still offers its image to the model first.
|
||||
await send([{ type: 'text', text: 'Look again.' }, image])
|
||||
expect(JSON.stringify(calls[1])).toContain('data:image/png;base64,kept-picture')
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch
|
||||
}
|
||||
})
|
||||
|
||||
test.each([
|
||||
'https://api.deepseek.com',
|
||||
'https://opencode.ai/zen/v1',
|
||||
'https://opencode.ai/zen/go/v1',
|
||||
'https://gateway.example.test/v1',
|
||||
].flatMap(baseUrl => ['deepseek-v4-pro', 'deepseek-flash', 'glm-5.3'].map(model => ({ baseUrl, model }))))(
|
||||
'offers images to $model at $baseUrl before any rejection',
|
||||
async ({ baseUrl, model }) => {
|
||||
// Guard: image support must never be predicted from host or model names.
|
||||
const body = await captureOpenAIChatRequest({
|
||||
baseUrl,
|
||||
model,
|
||||
contentSource: 'user',
|
||||
content: [
|
||||
{ type: 'text', text: 'Look.' },
|
||||
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'guard-picture' } },
|
||||
],
|
||||
})
|
||||
|
||||
expect(JSON.stringify(body)).toContain('data:image/png;base64,guard-picture')
|
||||
expect(JSON.stringify(body)).not.toContain('Image omitted:')
|
||||
},
|
||||
)
|
||||
|
||||
test('keeps offering images when the text-only resend fails for another reason', async () => {
|
||||
const originalFetch = globalThis.fetch
|
||||
const calls: Array<Record<string, unknown>> = []
|
||||
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
|
||||
const body = JSON.parse(String(init?.body)) as Record<string, unknown>
|
||||
calls.push(body)
|
||||
return JSON.stringify(body).includes('image_url')
|
||||
? Response.json({ error: { message: 'This model does not support image input' } }, { status: 400 })
|
||||
: Response.json({ error: { message: 'Rate limit reached' } }, { status: 429 })
|
||||
}) as typeof fetch
|
||||
|
||||
try {
|
||||
const svc = new ProviderService()
|
||||
const provider = await svc.addProvider(sampleInput({
|
||||
apiFormat: 'openai_chat',
|
||||
baseUrl: 'https://opencode.ai/zen/go/v1',
|
||||
models: { main: 'deepseek-flash', haiku: 'deepseek-flash', sonnet: 'deepseek-flash', opus: 'deepseek-flash' },
|
||||
}))
|
||||
await svc.activateProvider(provider.id)
|
||||
const send = () => {
|
||||
const req = new Request('http://localhost:3456/proxy/v1/messages', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: 'deepseek-flash',
|
||||
max_tokens: 64,
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: [
|
||||
{ type: 'text', text: 'Look.' },
|
||||
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'retry-picture' } },
|
||||
],
|
||||
}],
|
||||
}),
|
||||
})
|
||||
return handleProxyRequest(req, new URL(req.url))
|
||||
}
|
||||
|
||||
const first = await send()
|
||||
expect(first.status).toBe(429)
|
||||
expect(calls).toHaveLength(2)
|
||||
|
||||
await send()
|
||||
expect(JSON.stringify(calls[2])).toContain('data:image/png;base64,retry-picture')
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch
|
||||
}
|
||||
})
|
||||
|
||||
test('resends a streaming request without images after an image rejection', async () => {
|
||||
const originalFetch = globalThis.fetch
|
||||
const encoder = new TextEncoder()
|
||||
const calls: Array<Record<string, unknown>> = []
|
||||
globalThis.fetch = mock(async (_url: string | URL | Request, init?: RequestInit) => {
|
||||
const body = JSON.parse(String(init?.body)) as Record<string, unknown>
|
||||
calls.push(body)
|
||||
if (JSON.stringify(body).includes('image_url')) {
|
||||
return Response.json({
|
||||
error: { message: 'Error from provider (DeepSeek): unknown variant `image_url`, expected `text`' },
|
||||
}, { status: 400 })
|
||||
}
|
||||
return new Response(new ReadableStream<Uint8Array>({
|
||||
start(controller) {
|
||||
controller.enqueue(encoder.encode([
|
||||
'data: {"id":"chatcmpl-stream-retry","object":"chat.completion.chunk","model":"deepseek-flash","choices":[{"index":0,"delta":{"role":"assistant","content":"text answer"},"finish_reason":null}]}',
|
||||
'',
|
||||
'data: {"id":"chatcmpl-stream-retry","object":"chat.completion.chunk","model":"deepseek-flash","choices":[{"index":0,"delta":{},"finish_reason":"stop"}],"usage":{"prompt_tokens":5,"completion_tokens":2,"total_tokens":7}}',
|
||||
'',
|
||||
'data: [DONE]',
|
||||
'',
|
||||
].join('\n')))
|
||||
controller.close()
|
||||
},
|
||||
}), { status: 200, headers: { 'Content-Type': 'text/event-stream' } })
|
||||
}) as typeof fetch
|
||||
|
||||
try {
|
||||
const svc = new ProviderService()
|
||||
const provider = await svc.addProvider(sampleInput({
|
||||
apiFormat: 'openai_chat',
|
||||
baseUrl: 'https://opencode.ai/zen/go/v1',
|
||||
models: { main: 'deepseek-flash', haiku: 'deepseek-flash', sonnet: 'deepseek-flash', opus: 'deepseek-flash' },
|
||||
}))
|
||||
await svc.activateProvider(provider.id)
|
||||
const req = new Request('http://localhost:3456/proxy/v1/messages', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: 'deepseek-flash[1m]',
|
||||
max_tokens: 64,
|
||||
stream: true,
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: [
|
||||
{ type: 'text', text: 'Look.' },
|
||||
{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'stream-picture' } },
|
||||
],
|
||||
}],
|
||||
}),
|
||||
})
|
||||
|
||||
const res = await handleProxyRequest(req, new URL(req.url))
|
||||
expect(res.status).toBe(200)
|
||||
expect(res.headers.get('Content-Type')).toBe('text/event-stream')
|
||||
const text = await res.text()
|
||||
expect(text).toContain('text answer')
|
||||
expect(text).toContain('message_stop')
|
||||
expect(calls).toHaveLength(2)
|
||||
expect(calls[1].stream).toBe(true)
|
||||
expect(JSON.stringify(calls[1])).not.toContain('stream-picture')
|
||||
} finally {
|
||||
globalThis.fetch = originalFetch
|
||||
}
|
||||
})
|
||||
|
||||
test.each([
|
||||
@@ -2365,6 +2632,10 @@ describe('ProviderService', () => {
|
||||
{ model: 'space-bunny-free', contentSource: 'tool' },
|
||||
{ model: 'space-bunny-free[1m]', contentSource: 'user' },
|
||||
{ model: 'space-bunny-free[1m]', contentSource: 'tool' },
|
||||
{ model: 'deepseek-flash', contentSource: 'user' },
|
||||
{ model: 'deepseek-flash', contentSource: 'tool' },
|
||||
{ model: 'deepseek-flash[1m]', contentSource: 'user' },
|
||||
{ model: 'some-future-model', contentSource: 'tool' },
|
||||
] as const)('forwards OpenCode Go images for $model from $contentSource content', async ({ model, contentSource }) => {
|
||||
const body = await captureOpenAIChatRequest({
|
||||
baseUrl: 'https://opencode.ai/zen/go/v1',
|
||||
|
||||
@@ -865,7 +865,7 @@ describe('anthropicToOpenaiChat', () => {
|
||||
}
|
||||
const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' })
|
||||
expect(result.messages[0].content).toBe(
|
||||
'What is in this screenshot?\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n',
|
||||
'What is in this screenshot?\n[Image omitted: the upstream rejected image input for this model.]\n',
|
||||
)
|
||||
expect(JSON.stringify(result)).not.toContain('image_url')
|
||||
expect(JSON.stringify(result)).not.toContain('abc123')
|
||||
@@ -889,7 +889,7 @@ describe('anthropicToOpenaiChat', () => {
|
||||
}
|
||||
const result = anthropicToOpenaiChat(req, { imageContentMode: 'text_only' })
|
||||
expect(result.messages).toEqual([
|
||||
{ role: 'tool', tool_call_id: 'tc_img', content: '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n' },
|
||||
{ role: 'tool', tool_call_id: 'tc_img', content: '\n[Image omitted: the upstream rejected image input for this model.]\n' },
|
||||
])
|
||||
expect(JSON.stringify(result)).not.toContain('image_url')
|
||||
expect(JSON.stringify(result)).not.toContain('abc123')
|
||||
@@ -946,7 +946,7 @@ describe('anthropicToOpenaiChat', () => {
|
||||
expect(result.messages).toEqual([
|
||||
{
|
||||
role: 'user',
|
||||
content: '[Document: cited]\nbefore\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\nafter',
|
||||
content: '[Document: cited]\nbefore\n[Image omitted: the upstream rejected image input for this model.]\nafter',
|
||||
},
|
||||
])
|
||||
expect(JSON.stringify(result)).not.toContain('image_url')
|
||||
|
||||
+45
-26
@@ -15,9 +15,15 @@ import { normalizeAnthropicBaseUrl } from '../../services/api/anthropicBaseUrl.j
|
||||
import { createGunzip, createInflate } from 'node:zlib'
|
||||
|
||||
import { ProviderService } from '../services/providerService.js'
|
||||
import { diagnosticsService } from '../services/diagnosticsService.js'
|
||||
import type { ProviderAuthStrategy } from '../types/provider.js'
|
||||
import { resolvePromptCacheKey } from './promptCacheKey.js'
|
||||
import { anthropicToOpenaiChat } from './transform/anthropicToOpenaiChat.js'
|
||||
import {
|
||||
getOpenAIChatImageContentMode,
|
||||
isOpenAIChatImageRejection,
|
||||
rememberOpenAIChatTextOnlyModel,
|
||||
} from './openaiChatImageSupport.js'
|
||||
import { anthropicToOpenaiChat, type OpenAIChatImageContentMode } from './transform/anthropicToOpenaiChat.js'
|
||||
import { anthropicToOpenaiResponses } from './transform/anthropicToOpenaiResponses.js'
|
||||
import { RequestCompatibilityError, resolveRequestCompatibility, type RequestCompatibilityOptions } from './transform/requestCompatibility.js'
|
||||
import { ProtocolTraceObserver, observeProtocolStream, type ProtocolTraceTransport } from './protocolTrace.js'
|
||||
@@ -698,20 +704,22 @@ async function handleOpenaiChat(
|
||||
traceContext: ProxyTraceContext | null,
|
||||
requestOptions: RequestCompatibilityOptions = {},
|
||||
upstreamHeaders: Record<string, string> = {},
|
||||
imageContentModeOverride?: OpenAIChatImageContentMode,
|
||||
): Promise<Response> {
|
||||
const knownDeepSeekHost = shouldUseDeepSeekReasoningCompat(baseUrl)
|
||||
const reasoningProfile = resolveModelReasoningProfile(body.model, 'openai_chat')
|
||||
const url = buildOpenaiEndpoint(baseUrl, 'chat/completions')
|
||||
const imageContentMode = imageContentModeOverride ?? getOpenAIChatImageContentMode(url, body.model)
|
||||
const transformed = anthropicToOpenaiChat(body, {
|
||||
...requestOptions,
|
||||
roundTripReasoningContent: knownDeepSeekHost || reasoningProfile?.family === 'deepseek-v4',
|
||||
passThinkingToggle: knownDeepSeekHost,
|
||||
imageContentMode: shouldUseTextOnlyOpenAIChatContent(baseUrl, body.model) ? 'text_only' : 'vision',
|
||||
imageContentMode,
|
||||
})
|
||||
if (traceContext) {
|
||||
traceContext.protocolTrace = new ProtocolTraceObserver('openai_chat', transformed,
|
||||
resolveRequestCompatibility(body, { ...requestOptions, protocol: 'openai_chat' }).outputBudget)
|
||||
}
|
||||
const url = buildOpenaiEndpoint(baseUrl, 'chat/completions')
|
||||
// Preset-declared headers first: `Authorization` is applied last so a preset can
|
||||
// never shadow the credential, and the resolver already drops framing headers.
|
||||
const upstreamRequestHeaders: Record<string, string> = {}
|
||||
@@ -760,6 +768,40 @@ async function handleOpenaiChat(
|
||||
|
||||
if (!upstream.ok) {
|
||||
const errText = await upstream.text().catch(() => '')
|
||||
if (imageContentMode === 'vision' && isOpenAIChatImageRejection(upstream.status, errText, transformed)) {
|
||||
// The upstream refused the request before generating anything, so the
|
||||
// client has seen no output and resending is safe. The model is only
|
||||
// remembered once the same request succeeds without images, so an
|
||||
// unrelated failure of the resend cannot disable images for good.
|
||||
if (traceContext) {
|
||||
recordProxyTraceInBackground({
|
||||
context: traceContext,
|
||||
callId: traceCallId,
|
||||
model: body.model,
|
||||
upstreamUrl: url,
|
||||
upstreamRequest: transformed,
|
||||
requestHeaders: upstreamRequestHeaders,
|
||||
startedAt,
|
||||
startedAtMs,
|
||||
responseStatus: upstream.status,
|
||||
upstreamResponseBody: errText,
|
||||
responseHeaders: upstream.headers,
|
||||
})
|
||||
}
|
||||
const resent = await handleOpenaiChat(
|
||||
body, baseUrl, apiKey, isStream, networkSettings, traceContext, requestOptions, upstreamHeaders, 'text_only',
|
||||
)
|
||||
if (resent.ok) {
|
||||
rememberOpenAIChatTextOnlyModel(url, body.model)
|
||||
void diagnosticsService.recordEvent({
|
||||
type: 'openai_chat_image_input_disabled',
|
||||
severity: 'info',
|
||||
summary: `Upstream rejected image input for ${body.model}; images are omitted for this model for 30 minutes`,
|
||||
details: { model: body.model, upstreamUrl: url, httpStatus: upstream.status, upstreamError: errText.slice(0, 500) },
|
||||
})
|
||||
}
|
||||
return resent
|
||||
}
|
||||
let policyError = null
|
||||
try {
|
||||
policyError = getOpenAIPolicyError(JSON.parse(errText))
|
||||
@@ -880,29 +922,6 @@ function shouldUseDeepSeekReasoningCompat(baseUrl: string): boolean {
|
||||
)
|
||||
}
|
||||
|
||||
function shouldUseTextOnlyOpenAIChatContent(baseUrl: string, model: string): boolean {
|
||||
// Keep classic DeepSeek text models compatible without dropping images for
|
||||
// explicitly vision-capable models served by the same Chat endpoint.
|
||||
if (/(^|[./-])deepseek([./-]|$)/i.test(baseUrl)) {
|
||||
return !hasExplicitVisionModelMarker(model)
|
||||
}
|
||||
|
||||
// OpenCode Go's Kimi K3 and Space Bunny Free accept image_url despite lacking
|
||||
// "vision" in their model ids. Keep other unverified gateway models on the
|
||||
// text-only path. Context-window suffixes have already been stripped.
|
||||
if (/(^|[./-])opencode\.ai([:/]|$)/i.test(baseUrl)) {
|
||||
return !hasExplicitVisionModelMarker(model) && !['kimi-k3', 'space-bunny-free'].includes(model.toLowerCase())
|
||||
}
|
||||
|
||||
// Preserve the existing behavior for generic compatible providers whose
|
||||
// capabilities are not controlled by either compatibility policy above.
|
||||
return false
|
||||
}
|
||||
|
||||
function hasExplicitVisionModelMarker(model: string): boolean {
|
||||
return /(^|[/:._-])vision([/:._-]|$)/i.test(model)
|
||||
}
|
||||
|
||||
async function handleOpenaiResponses(
|
||||
body: AnthropicRequest,
|
||||
baseUrl: string,
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
import { afterEach, describe, expect, test } from 'bun:test'
|
||||
import {
|
||||
getOpenAIChatImageContentMode,
|
||||
isOpenAIChatImageRejection,
|
||||
rememberOpenAIChatTextOnlyModel,
|
||||
resetOpenAIChatImageSupportForTests,
|
||||
} from './openaiChatImageSupport.js'
|
||||
import type { OpenAIChatRequest } from './transform/types.js'
|
||||
|
||||
const withImage: OpenAIChatRequest = {
|
||||
model: 'deepseek-flash',
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: [
|
||||
{ type: 'text', text: 'Look.' },
|
||||
{ type: 'image_url', image_url: { url: 'data:image/png;base64,AAAA' } },
|
||||
],
|
||||
}],
|
||||
}
|
||||
const withoutImage: OpenAIChatRequest = { model: 'deepseek-flash', messages: [{ role: 'user', content: 'Look.' }] }
|
||||
|
||||
afterEach(resetOpenAIChatImageSupportForTests)
|
||||
|
||||
describe('isOpenAIChatImageRejection', () => {
|
||||
test('accepts an explicit refusal of image input', () => {
|
||||
const refusals = [
|
||||
'Error from provider (DeepSeek): Failed to deserialize the JSON body into the target type: messages[0]: unknown variant `image_url`, expected `text` at line 1 column 120',
|
||||
'Error from provider (Moonshot): This model does not support image input',
|
||||
"messages.0.content.1.type: Input should be 'text'; received 'image_url'",
|
||||
'unsupported modality: image input is not available',
|
||||
'Qwen/Qwen3-8B is not a multimodal model',
|
||||
'Invalid content type. image_url is only supported by certain models.',
|
||||
]
|
||||
|
||||
for (const refusal of refusals) {
|
||||
expect(isOpenAIChatImageRejection(400, refusal, withImage)).toBe(true)
|
||||
}
|
||||
expect(isOpenAIChatImageRejection(422, refusals[2]!, withImage)).toBe(true)
|
||||
})
|
||||
|
||||
test('rejects failures about one particular image', () => {
|
||||
const imageProblems = [
|
||||
"You uploaded an unsupported image. Please make sure your image has of one the following formats: ['png', 'jpeg', 'gif', 'webp'].",
|
||||
"Invalid image_url: unsupported MIME type 'image/heic'",
|
||||
'unsupported image format: image/bmp',
|
||||
'Image format not supported',
|
||||
'Image is not supported: exceeds the 20 MB limit',
|
||||
'image_url is not supported: failed to download the image',
|
||||
'Unsupported image: could not decode the image data',
|
||||
'Multiple images are not supported by this model',
|
||||
'Animated images are not supported',
|
||||
'Image type gif is not supported',
|
||||
'Image input is not supported in your region',
|
||||
]
|
||||
|
||||
for (const problem of imageProblems) {
|
||||
expect(isOpenAIChatImageRejection(400, problem, withImage)).toBe(false)
|
||||
}
|
||||
})
|
||||
|
||||
test('rejects server failures and requests without images', () => {
|
||||
const refusal = 'This model does not support image input'
|
||||
expect(isOpenAIChatImageRejection(500, refusal, withImage)).toBe(false)
|
||||
expect(isOpenAIChatImageRejection(429, refusal, withImage)).toBe(false)
|
||||
expect(isOpenAIChatImageRejection(400, refusal, withoutImage)).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('learned text-only models', () => {
|
||||
test('are scoped to one endpoint and model, ignoring case and trailing slashes', () => {
|
||||
rememberOpenAIChatTextOnlyModel('https://opencode.ai/zen/go/v1/chat/completions/', 'DeepSeek-V4-Pro')
|
||||
|
||||
expect(getOpenAIChatImageContentMode('https://opencode.ai/zen/go/v1/chat/completions', 'deepseek-v4-pro')).toBe('text_only')
|
||||
expect(getOpenAIChatImageContentMode('https://opencode.ai/zen/go/v1/chat/completions', 'deepseek-flash')).toBe('vision')
|
||||
expect(getOpenAIChatImageContentMode('https://api.deepseek.com/chat/completions', 'deepseek-v4-pro')).toBe('vision')
|
||||
})
|
||||
|
||||
test('expire after 30 minutes so the model is probed with images again', () => {
|
||||
const endpoint = 'https://opencode.ai/zen/go/v1/chat/completions'
|
||||
const learnedAt = 1_000_000
|
||||
rememberOpenAIChatTextOnlyModel(endpoint, 'deepseek-v4-pro', learnedAt)
|
||||
|
||||
expect(getOpenAIChatImageContentMode(endpoint, 'deepseek-v4-pro', learnedAt + 29 * 60_000)).toBe('text_only')
|
||||
expect(getOpenAIChatImageContentMode(endpoint, 'deepseek-v4-pro', learnedAt + 31 * 60_000)).toBe('vision')
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,80 @@
|
||||
import { isUnsupportedImageInputErrorMessage } from '../../services/api/unsupportedImageInput.js'
|
||||
import type { OpenAIChatImageContentMode } from './transform/anthropicToOpenaiChat.js'
|
||||
import type { OpenAIChatRequest } from './transform/types.js'
|
||||
|
||||
// Image support is learned from the upstream, not predicted from host or model
|
||||
// names: gateways such as OpenCode Go add vision models faster than any static
|
||||
// list can follow. Every model starts on the vision path; a model is moved to
|
||||
// text-only only after its endpoint explicitly rejects image input and the
|
||||
// same request succeeds without images. The knowledge expires, so a model that
|
||||
// gains vision support (or a rare misreading) recovers without a restart.
|
||||
const TEXT_ONLY_TTL_MS = 30 * 60 * 1000
|
||||
const MAX_TEXT_ONLY_MODELS = 256
|
||||
const textOnlyModelsUntil = new Map<string, number>()
|
||||
|
||||
// Failures about particular images (format, MIME type, size, animation, image
|
||||
// count, broken data, unreachable URL) also mention images, but the model
|
||||
// still accepts other images. Learning from them would strip every later image
|
||||
// for this model, so they surface unchanged; a missed refusal only costs one
|
||||
// visible error.
|
||||
const IMAGE_CONTENT_PROBLEM = new RegExp([
|
||||
String.raw`\bmime\b`, String.raw`\bformats?\b`, String.raw`\bgif\b`, 'animated', 'image type',
|
||||
'multiple images', 'too many images', 'number of images',
|
||||
'too large', 'exceed', String.raw`\bsize\b`, 'dimension', 'resolution', 'pixel',
|
||||
'corrupt', 'decod', 'base64', 'download', 'fetch', 'could not process', 'unable to process', 'region',
|
||||
].join('|'), 'i')
|
||||
|
||||
function modelKey(endpointUrl: string, model: string): string {
|
||||
return `${endpointUrl.replace(/\/+$/, '').toLowerCase()}\n${model.toLowerCase()}`
|
||||
}
|
||||
|
||||
export function getOpenAIChatImageContentMode(
|
||||
endpointUrl: string,
|
||||
model: string,
|
||||
now = Date.now(),
|
||||
): OpenAIChatImageContentMode {
|
||||
const key = modelKey(endpointUrl, model)
|
||||
const until = textOnlyModelsUntil.get(key)
|
||||
if (until === undefined) return 'vision'
|
||||
if (until > now) return 'text_only'
|
||||
textOnlyModelsUntil.delete(key)
|
||||
return 'vision'
|
||||
}
|
||||
|
||||
export function rememberOpenAIChatTextOnlyModel(endpointUrl: string, model: string, now = Date.now()): void {
|
||||
const key = modelKey(endpointUrl, model)
|
||||
textOnlyModelsUntil.delete(key)
|
||||
if (textOnlyModelsUntil.size >= MAX_TEXT_ONLY_MODELS) {
|
||||
const oldest = textOnlyModelsUntil.keys().next().value
|
||||
if (oldest !== undefined) textOnlyModelsUntil.delete(oldest)
|
||||
}
|
||||
textOnlyModelsUntil.set(key, now + TEXT_ONLY_TTL_MS)
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a failed upstream response is the endpoint refusing image input for
|
||||
* this model. Only an explicit refusal of a request that actually carried
|
||||
* images qualifies: other failures (a bad image, context overflow, 5xx) must
|
||||
* surface unchanged instead of silently dropping the user's images.
|
||||
*/
|
||||
export function isOpenAIChatImageRejection(
|
||||
status: number,
|
||||
errorText: string,
|
||||
request: OpenAIChatRequest,
|
||||
): boolean {
|
||||
return (
|
||||
(status === 400 || status === 422) &&
|
||||
requestCarriesImages(request) &&
|
||||
isUnsupportedImageInputErrorMessage(errorText) &&
|
||||
!IMAGE_CONTENT_PROBLEM.test(errorText)
|
||||
)
|
||||
}
|
||||
|
||||
function requestCarriesImages(request: OpenAIChatRequest): boolean {
|
||||
return request.messages.some(message =>
|
||||
Array.isArray(message.content) && message.content.some(part => part.type === 'image_url'))
|
||||
}
|
||||
|
||||
export function resetOpenAIChatImageSupportForTests(): void {
|
||||
textOnlyModelsUntil.clear()
|
||||
}
|
||||
@@ -18,7 +18,7 @@ import { stripLeadingBillingHeader } from './billingHeader.js'
|
||||
import { normalizeOpenAIReasoningEffort } from './effort.js'
|
||||
import { resolveRequestCompatibility, type RequestCompatibilityOptions } from './requestCompatibility.js'
|
||||
|
||||
type OpenAIChatImageContentMode = 'vision' | 'text_only'
|
||||
export type OpenAIChatImageContentMode = 'vision' | 'text_only'
|
||||
|
||||
// Synthetic text parts (degraded documents, search results) carry an internal
|
||||
// marker so serializers can preserve their boundaries. The marker is removed
|
||||
@@ -35,7 +35,7 @@ export type OpenAIChatTransformOptions = RequestCompatibilityOptions & {
|
||||
// Synthetic degradation text carries its own separators: the parts are
|
||||
// joined without a separator, so a notice must not glue itself to the
|
||||
// surrounding user text.
|
||||
const OMITTED_IMAGE_TEXT = '\n[Image omitted: this OpenAI-compatible chat endpoint only supports text content.]\n'
|
||||
const OMITTED_IMAGE_TEXT = '\n[Image omitted: the upstream rejected image input for this model.]\n'
|
||||
const FILE_IMAGE_OMITTED_TEXT = '\n[Image omitted: file-based image source is not supported by this endpoint.]\n'
|
||||
const MEDIA_RESULT_ATTACHED_TEXT = 'Media result attached after this tool result.'
|
||||
const DOCUMENT_TEXT_INLINE_LIMIT = 2000
|
||||
|
||||
@@ -56,6 +56,9 @@ import {
|
||||
StreamEndedEarlyError,
|
||||
} from './streamFallback.js'
|
||||
import { StreamWatchdogTimeoutError } from './streamWatchdog.js'
|
||||
import { isUnsupportedImageInputErrorMessage } from './unsupportedImageInput.js'
|
||||
|
||||
export { isUnsupportedImageInputErrorMessage }
|
||||
|
||||
// Presentation only: classifiers, retries and diagnostic metadata keep the
|
||||
// original SDK error. Decode envelopes structurally so escaped quotes/newlines
|
||||
@@ -596,48 +599,6 @@ export function extractUnknownErrorFormat(value: unknown): string | undefined {
|
||||
return undefined
|
||||
}
|
||||
|
||||
export function isUnsupportedImageInputErrorMessage(message: string): boolean {
|
||||
const raw = message.toLowerCase()
|
||||
if (!raw.includes('image')) return false
|
||||
if (isOpenAIImageUrlTextOnlySchemaError(raw)) {
|
||||
return true
|
||||
}
|
||||
return (
|
||||
raw.includes('not support') ||
|
||||
raw.includes('not supported') ||
|
||||
raw.includes('unsupported') ||
|
||||
raw.includes('vision') ||
|
||||
raw.includes('multimodal') ||
|
||||
raw.includes('multi-modal') ||
|
||||
raw.includes('modality')
|
||||
)
|
||||
}
|
||||
|
||||
function isOpenAIImageUrlTextOnlySchemaError(raw: string): boolean {
|
||||
if (!raw.includes('image_url')) return false
|
||||
if (
|
||||
raw.includes('not allowed') ||
|
||||
raw.includes('not permitted') ||
|
||||
raw.includes('disallowed') ||
|
||||
raw.includes('forbidden')
|
||||
) {
|
||||
return true
|
||||
}
|
||||
if (!raw.includes('text')) return false
|
||||
return (
|
||||
raw.includes('expected') ||
|
||||
raw.includes('input should be') ||
|
||||
raw.includes('not one of') ||
|
||||
raw.includes('permitted') ||
|
||||
raw.includes('received') ||
|
||||
raw.includes('unknown variant') ||
|
||||
raw.includes('invalid value') ||
|
||||
raw.includes('invalid type') ||
|
||||
raw.includes('valid enumeration') ||
|
||||
raw.includes('only text')
|
||||
)
|
||||
}
|
||||
|
||||
// Deep-walk a request payload looking for image blocks. Images can sit at the
|
||||
// top level of a user message or nested inside a tool_result's content, and
|
||||
// the payloads here are { type, message: { content } } wrappers, so walk all
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
import { describe, expect, test } from 'bun:test'
|
||||
import { isUnsupportedImageInputErrorMessage } from './unsupportedImageInput.js'
|
||||
|
||||
describe('isUnsupportedImageInputErrorMessage', () => {
|
||||
test('recognizes rejections relayed through a gateway prefix', () => {
|
||||
// OpenCode Zen/Go wraps the upstream message as "Error from provider (<name>): <message>".
|
||||
const relayed = [
|
||||
'Error from provider (DeepSeek): Failed to deserialize the JSON body into the target type: messages[2]: unknown variant `image_url`, expected `text` at line 1 column 512',
|
||||
'Error from provider (Moonshot): This model does not support image input',
|
||||
'{"error":{"message":"Error from provider: Invalid value for \'messages[0].content[1].type\': \'image_url\' is not one of [\'text\']"}}',
|
||||
// vLLM names the modality instead of images.
|
||||
'Qwen/Qwen3-8B is not a multimodal model',
|
||||
'Invalid content type. image_url is only supported by certain models.',
|
||||
]
|
||||
|
||||
for (const message of relayed) {
|
||||
expect(isUnsupportedImageInputErrorMessage(message)).toBe(true)
|
||||
}
|
||||
})
|
||||
|
||||
test('ignores image failures that are not about the model refusing images', () => {
|
||||
const unrelated = [
|
||||
'image exceeds maximum',
|
||||
'Error from provider (DeepSeek): Image is too large: 34 MB exceeds the 20 MB limit',
|
||||
'Error from provider (DeepSeek): This model maximum context length is 131072 tokens',
|
||||
'unsupported content block type: only text is allowed for this model',
|
||||
]
|
||||
|
||||
for (const message of unrelated) {
|
||||
expect(isUnsupportedImageInputErrorMessage(message)).toBe(false)
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,50 @@
|
||||
/**
|
||||
* Recognizes provider errors that reject image input for a text-only model.
|
||||
*
|
||||
* Kept free of CLI state so both the CLI error mapper and the desktop proxy
|
||||
* (which retries OpenAI Chat requests without images) share one definition.
|
||||
*/
|
||||
export function isUnsupportedImageInputErrorMessage(message: string): boolean {
|
||||
const raw = message.toLowerCase()
|
||||
// vLLM: "<model> is not a multimodal model" names the modality, not images.
|
||||
if (/\bnot an? (multimodal|multi-modal|vision) model\b/.test(raw)) return true
|
||||
if (!raw.includes('image')) return false
|
||||
if (isOpenAIImageUrlTextOnlySchemaError(raw)) {
|
||||
return true
|
||||
}
|
||||
return (
|
||||
raw.includes('not support') ||
|
||||
raw.includes('not supported') ||
|
||||
raw.includes('unsupported') ||
|
||||
raw.includes('vision') ||
|
||||
raw.includes('multimodal') ||
|
||||
raw.includes('multi-modal') ||
|
||||
raw.includes('modality')
|
||||
)
|
||||
}
|
||||
|
||||
function isOpenAIImageUrlTextOnlySchemaError(raw: string): boolean {
|
||||
if (!raw.includes('image_url')) return false
|
||||
if (
|
||||
raw.includes('not allowed') ||
|
||||
raw.includes('not permitted') ||
|
||||
raw.includes('disallowed') ||
|
||||
raw.includes('forbidden') ||
|
||||
raw.includes('only supported by certain models')
|
||||
) {
|
||||
return true
|
||||
}
|
||||
if (!raw.includes('text')) return false
|
||||
return (
|
||||
raw.includes('expected') ||
|
||||
raw.includes('input should be') ||
|
||||
raw.includes('not one of') ||
|
||||
raw.includes('permitted') ||
|
||||
raw.includes('received') ||
|
||||
raw.includes('unknown variant') ||
|
||||
raw.includes('invalid value') ||
|
||||
raw.includes('invalid type') ||
|
||||
raw.includes('valid enumeration') ||
|
||||
raw.includes('only text')
|
||||
)
|
||||
}
|
||||
Reference in New Issue
Block a user