diff --git a/src/server/__tests__/proxy-transform.test.ts b/src/server/__tests__/proxy-transform.test.ts index 2af6ea08..71cada78 100644 --- a/src/server/__tests__/proxy-transform.test.ts +++ b/src/server/__tests__/proxy-transform.test.ts @@ -621,6 +621,66 @@ describe('anthropicToOpenaiResponses', () => { expect((result as Record).stop).toBeUndefined() expect((result as Record).stop_sequences).toBeUndefined() }) + + // Responses names the function inline. The nested {function:{name}} shape is + // Chat Completions syntax and strict upstreams (xAI) reject it. + test('tool_choice type=tool names the function inline, not nested', () => { + const req: AnthropicRequest = { + model: 'gpt-4o', + max_tokens: 100, + messages: [{ role: 'user', content: 'Hi' }], + tools: [{ name: 'get_weather', input_schema: { type: 'object' } }], + tool_choice: { type: 'tool', name: 'get_weather' }, + } + const result = anthropicToOpenaiResponses(req) + expect(result.tool_choice).toEqual({ type: 'function', name: 'get_weather' }) + }) + + test('drops tool_choice when the request carries no tools', () => { + for (const choice of [ + { type: 'auto' }, + { type: 'any' }, + { type: 'tool', name: 'get_weather' }, + ]) { + const result = anthropicToOpenaiResponses({ + model: 'gpt-4o', + max_tokens: 100, + messages: [{ role: 'user', content: 'Hi' }], + tool_choice: choice, + } as AnthropicRequest) + expect(result.tools).toBeUndefined() + expect(result.tool_choice).toBeUndefined() + } + }) + + test('drops a tool_choice orphaned by tool filtering', () => { + const result = anthropicToOpenaiResponses({ + model: 'gpt-4o', + max_tokens: 100, + messages: [{ role: 'user', content: 'Hi' }], + // BatchTool is filtered out of the tool list, so a choice naming it + // would point at a tool the upstream never receives. + tools: [{ name: 'BatchTool', input_schema: { type: 'object' } }], + tool_choice: { type: 'tool', name: 'BatchTool' }, + } as AnthropicRequest) + expect(result.tools).toBeUndefined() + expect(result.tool_choice).toBeUndefined() + }) + + test('keeps a tool_choice whose target survives filtering', () => { + const result = anthropicToOpenaiResponses({ + model: 'gpt-4o', + max_tokens: 100, + messages: [{ role: 'user', content: 'Hi' }], + tools: [ + { name: 'BatchTool', input_schema: { type: 'object' } }, + { name: 'get_weather', input_schema: { type: 'object' } }, + ], + tool_choice: { type: 'tool', name: 'get_weather' }, + } as AnthropicRequest) + expect(result.tools).toHaveLength(1) + expect(result.tool_choice).toEqual({ type: 'function', name: 'get_weather' }) + }) }) // ─── openaiResponsesToAnthropic ───────────────────────────────── diff --git a/src/server/proxy/transform/anthropicToOpenaiResponses.ts b/src/server/proxy/transform/anthropicToOpenaiResponses.ts index 12648fc6..2a1a0176 100644 --- a/src/server/proxy/transform/anthropicToOpenaiResponses.ts +++ b/src/server/proxy/transform/anthropicToOpenaiResponses.ts @@ -68,9 +68,10 @@ export function anthropicToOpenaiResponses( if (body.top_p !== undefined) result.top_p = body.top_p } - // tools + // tools — an empty array after filtering is dropped, not sent as `[]`, so it + // reads as "no tools" to strict upstreams instead of "an empty tool set". if (body.tools && body.tools.length > 0) { - result.tools = body.tools + const tools = body.tools .filter((t) => t.name !== 'BatchTool') .map((t) => ({ type: 'function', @@ -78,11 +79,20 @@ export function anthropicToOpenaiResponses( description: t.description, parameters: t.input_schema, })) + if (tools.length > 0) { + result.tools = tools + } } - // tool_choice + // tool_choice — only meaningful next to the tools it selects from. A choice + // that outlives its tool (BatchTool filtered above, or a client that sends + // tool_choice with no tools at all) is an orphan that strict Responses + // upstreams reject. if (body.tool_choice !== undefined) { - result.tool_choice = convertToolChoice(body.tool_choice) + const toolChoice = convertToolChoice(body.tool_choice) + if (isSelectableToolChoice(toolChoice, result.tools)) { + result.tool_choice = toolChoice + } } // thinking → reasoning @@ -184,8 +194,27 @@ function convertToolChoice(choice: unknown): unknown { if (c.type === 'any') return 'required' if (c.type === 'none') return 'none' if (c.type === 'tool' && typeof c.name === 'string') { - return { type: 'function', function: { name: c.name } } + // Responses names the function inline: {type:'function', name}. The + // nested {function:{name}} form belongs to Chat Completions and is + // rejected here (see anthropicToOpenaiChat for that shape). + return { type: 'function', name: c.name } } } return 'auto' } + +/** + * A named tool_choice is only valid while its target survives into the request. + * Anything else — a choice with no tools at all — is dropped so the upstream + * never sees a selector pointing at nothing. + */ +function isSelectableToolChoice( + choice: unknown, + tools: { name: string }[] | undefined, +): boolean { + if (!tools || tools.length === 0) return false + if (typeof choice !== 'object' || choice === null) return true + const name = (choice as Record).name + if (typeof name !== 'string') return true + return tools.some((tool) => tool.name === name) +} diff --git a/src/services/grokAuth/fetch.test.ts b/src/services/grokAuth/fetch.test.ts index 0bb9416a..b3919b77 100644 --- a/src/services/grokAuth/fetch.test.ts +++ b/src/services/grokAuth/fetch.test.ts @@ -9,6 +9,7 @@ import { } from './fetch.js' import { GROK_OAUTH_FILE_ENV_KEY } from './storage.js' import { GROK_OAUTH_TOKEN_ENDPOINT } from './client.js' +import { isRetryableStreamTransportError } from '../api/withRetry.js' describe('Grok Responses fetch adapter', () => { let tmpDir: string @@ -66,6 +67,7 @@ describe('Grok Responses fetch adapter', () => { expect(call?.headers.get('Authorization')).toBe('Bearer access') expect(call?.headers.get('X-XAI-Token-Auth')).toBe('xai-grok-cli') expect(call?.headers.get('x-grok-client-version')).toBe(GROK_CLI_VERSION) + expect(call?.headers.get('x-grok-client-mode')).toBe('interactive') expect(call?.headers.get('User-Agent')).toBe(`xai-grok-workspace/${GROK_CLI_VERSION}`) expect(call?.headers.get('x-grok-model-override')).toBe('grok-4.5') expect(call?.body.model).toBe('grok-4.5') @@ -167,6 +169,171 @@ describe('Grok Responses fetch adapter', () => { }) }) + // Every turn of one conversation must land on the same cache entry, or the + // whole prefix is re-billed each time. xAI reads the identity from the body, + // the CLI proxy from a header — both must carry it, and both must survive the + // 401 refresh retry. + describe('prompt cache identity', () => { + async function capture( + body: Record, + init: RequestInit = {}, + ): Promise<{ headers: Headers; body: Record }[]> { + const calls: { headers: Headers; body: Record }[] = [] + const fetchOverride: typeof fetch = async (input, requestInit) => { + if (String(input) === GROK_OAUTH_TOKEN_ENDPOINT) { + return Response.json({ + access_token: 'new-access', + refresh_token: 'new-refresh', + expires_in: 3600, + }) + } + calls.push({ + headers: new Headers(requestInit?.headers), + body: JSON.parse(String(requestInit?.body)), + }) + return new Response( + 'event: response.completed\ndata: {"response":{"id":"r","object":"response","created_at":1,"model":"grok-4.5","status":"completed","output":[],"usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}\n\n', + { headers: { 'Content-Type': 'text/event-stream' } }, + ) + } + await buildGrokFetch(fetchOverride, 'test')( + 'https://api.anthropic.com/v1/messages', + { method: 'POST', ...init, body: JSON.stringify(body) }, + ) + return calls + } + + const anthropicBody = (metadata?: Record) => ({ + model: 'grok-4.5', + max_tokens: 64, + messages: [{ role: 'user', content: 'hello' }], + ...(metadata ? { metadata } : {}), + }) + + test("derives it from Claude Code's session-suffixed user_id", async () => { + const calls = await capture( + anthropicBody({ user_id: 'user_abc_session_sess-123' }), + ) + expect(calls[0]?.body.prompt_cache_key).toBe('sess-123') + expect(calls[0]?.headers.get('x-grok-conv-id')).toBe('sess-123') + }) + + test('falls back to the CLI session header', async () => { + const calls = await capture(anthropicBody(), { + headers: { 'X-Claude-Code-Session-Id': 'sess-header' }, + }) + expect(calls[0]?.body.prompt_cache_key).toBe('sess-header') + expect(calls[0]?.headers.get('x-grok-conv-id')).toBe('sess-header') + }) + + test('sends no identity rather than an unstable one', async () => { + const calls = await capture(anthropicBody()) + expect(calls[0]?.body.prompt_cache_key).toBeUndefined() + expect(calls[0]?.headers.get('x-grok-conv-id')).toBeNull() + }) + + test('keeps the identity on the retry after a 401 refresh', async () => { + const calls: { headers: Headers; body: Record }[] = [] + const fetchOverride: typeof fetch = async (input, requestInit) => { + if (String(input) === GROK_OAUTH_TOKEN_ENDPOINT) { + return Response.json({ + access_token: 'new-access', + refresh_token: 'new-refresh', + expires_in: 3600, + }) + } + calls.push({ + headers: new Headers(requestInit?.headers), + body: JSON.parse(String(requestInit?.body)), + }) + if (calls.length === 1) return new Response('expired', { status: 401 }) + return new Response( + 'event: response.completed\ndata: {"response":{"id":"r","object":"response","created_at":1,"model":"grok-4.5","status":"completed","output":[],"usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}\n\n', + { headers: { 'Content-Type': 'text/event-stream' } }, + ) + } + await buildGrokFetch(fetchOverride, 'test')( + 'https://api.anthropic.com/v1/messages', + { + method: 'POST', + body: JSON.stringify( + anthropicBody({ user_id: 'user_abc_session_sess-retry' }), + ), + }, + ) + + expect(calls).toHaveLength(2) + for (const call of calls) { + expect(call.headers.get('x-grok-conv-id')).toBe('sess-retry') + expect(call.headers.get('x-grok-model-override')).toBe('grok-4.5') + expect(call.headers.get('x-grok-client-mode')).toBe('interactive') + expect(call.body.prompt_cache_key).toBe('sess-retry') + } + }) + }) + + // The Grok subscription endpoint is reached over a long-lived TLS stream held + // open by the CLI process, so a proxy/NAT/edge reset lands mid-response + // rather than on stream creation. Drive a real socket reset — not a fake + // error object — so the classifier is pinned to what the runtime actually + // throws through the SSE transform. + test('surfaces a mid-stream socket reset as a retryable transport error', async () => { + const upstream = Bun.listen({ + hostname: '127.0.0.1', + port: 0, + socket: { + data(socket) { + socket.write( + 'HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nTransfer-Encoding: chunked\r\n\r\n', + ) + const event = [ + 'event: response.created', + 'data: {"model":"grok-4.5"}', + '', + '', + ].join('\n') + socket.write(`${event.length.toString(16)}\r\n${event}\r\n`) + setTimeout(() => socket.terminate(), 20) + }, + }, + }) + + try { + const response = await buildGrokFetch( + (_input, init) => fetch(`http://127.0.0.1:${upstream.port}/v1/responses`, init), + 'test', + )('https://api.anthropic.com/v1/messages', { + method: 'POST', + body: JSON.stringify({ + model: 'grok-4.5', + max_tokens: 64, + stream: true, + messages: [{ role: 'user', content: 'hello' }], + }), + }) + + expect(response.status).toBe(200) + + let thrown: unknown + try { + const reader = response.body!.getReader() + for (;;) { + const { done } = await reader.read() + if (done) break + } + } catch (error) { + thrown = error + } + + // The transform must propagate the fault, not close the stream cleanly — + // a clean close would look like a finished (truncated) turn. + expect(thrown).toBeDefined() + expect(isRetryableStreamTransportError(thrown)).toBe(true) + } finally { + upstream.stop(true) + } + }) + test('does not refresh-loop on entitlement failures', async () => { let calls = 0 const response = await buildGrokFetch(async () => { diff --git a/src/services/grokAuth/fetch.ts b/src/services/grokAuth/fetch.ts index 4d12a1c6..e8e056e8 100644 --- a/src/services/grokAuth/fetch.ts +++ b/src/services/grokAuth/fetch.ts @@ -1,3 +1,4 @@ +import { resolvePromptCacheKey } from '../../server/proxy/promptCacheKey.js' import { anthropicToOpenaiResponses } from '../../server/proxy/transform/anthropicToOpenaiResponses.js' import { openaiResponsesStreamToAnthropic } from '../../server/proxy/streaming/openaiResponsesStreamToAnthropic.js' import { openaiResponsesStreamToAnthropicResponse } from '../../server/proxy/streaming/openaiResponsesStreamToAnthropicResponse.js' @@ -22,6 +23,9 @@ export function buildGrokIdentityHeaders(accessToken: string): Headers { 'Content-Type': 'application/json', 'X-XAI-Token-Auth': 'xai-grok-cli', 'x-grok-client-version': GROK_CLI_VERSION, + // The CLI gateway distinguishes interactive sessions from batch traffic; + // the official client always declares one. + 'x-grok-client-mode': 'interactive', 'User-Agent': `xai-grok-workspace/${GROK_CLI_VERSION}`, }) } @@ -38,10 +42,14 @@ export function buildGrokFetch( const originalBody = await readAnthropicBody(input, init) const requestedModel = resolveGrokModel(originalBody.model) - const transformedBody = anthropicToOpenaiResponses({ - ...originalBody, - model: requestedModel, - }) + // One conversation must route to one cache entry, or every turn re-bills + // the whole prefix. xAI reads it from the body; the CLI proxy also keys its + // conversation state off a header, so both carry the same identity. + const cacheKey = resolvePromptCacheKey(originalBody, readSessionId(input, init)) + const transformedBody = anthropicToOpenaiResponses( + { ...originalBody, model: requestedModel }, + cacheKey ? { cacheKey } : {}, + ) transformedBody.model = requestedModel transformedBody.stream = true if (grokModelRejectsReasoningEffort(requestedModel)) { @@ -54,8 +62,14 @@ export function buildGrokFetch( 'Grok OAuth token is missing or expired. Authorize Grok again in the desktop app.', ) } - const headers = buildGrokIdentityHeaders(tokens.accessToken) - headers.set('x-grok-model-override', requestedModel) + const applyRequestIdentity = (requestHeaders: Headers): Headers => { + requestHeaders.set('x-grok-model-override', requestedModel) + if (cacheKey) requestHeaders.set('x-grok-conv-id', cacheKey) + return requestHeaders + } + const headers = applyRequestIdentity( + buildGrokIdentityHeaders(tokens.accessToken), + ) void source const requestUpstream = (requestHeaders: Headers) => inner(GROK_CLI_API_ENDPOINT, { @@ -69,9 +83,9 @@ export function buildGrokFetch( if (upstream.status === 401) { const refreshed = await forceRefreshGrokTokens({ fetchOverride: inner }) if (refreshed) { - const refreshedHeaders = buildGrokIdentityHeaders(refreshed.accessToken) - refreshedHeaders.set('x-grok-model-override', requestedModel) - upstream = await requestUpstream(refreshedHeaders) + upstream = await requestUpstream( + applyRequestIdentity(buildGrokIdentityHeaders(refreshed.accessToken)), + ) } } @@ -118,6 +132,22 @@ export function buildGrokFetch( } } +/** + * The CLI stamps its session id on every request as a default header, which is + * the last-resort cache identity when the body carries no session metadata. + */ +function readSessionId( + input: RequestInfo | URL, + init?: RequestInit, +): string | null { + const header = 'x-claude-code-session-id' + if (init?.headers) { + const fromInit = new Headers(init.headers).get(header) + if (fromInit) return fromInit + } + return input instanceof Request ? input.headers.get(header) : null +} + async function readAnthropicBody( input: RequestInfo | URL, init?: RequestInit,