fix(grok): align the Responses request with what xAI actually accepts

Four gaps against the official Grok CLI surface, found by comparing this
adapter with sub2api's and each confirmed by a probe against the live
endpoint.

The first is not a parity gap but a hard failure. A named tool_choice was
translated into Chat Completions syntax, `{type:'function',function:{name}}`,
which the Responses endpoint refuses:

    422 data did not match any variant of untagged enum ModelToolChoice

so every request where the caller pins a tool failed outright. Responses
names the function inline. Only the Responses transform changes; the Chat
transform keeps the nested form, which is correct there.

The rest:

- A tool_choice sent with no tools, or one whose target the BatchTool
  filter removed, is dropped rather than left selecting a tool the upstream
  never receives — and a tool list that filters down to empty is omitted
  instead of sent as `[]`.
- `prompt_cache_key` and the `x-grok-conv-id` header carry one identity per
  conversation, reusing resolvePromptCacheKey (the Claude Code session id)
  so a turn stops re-billing the whole prefix. Both survive the 401 refresh
  retry: header assembly is folded into one helper so the first attempt and
  the retry cannot drift apart.
- `x-grok-client-mode: interactive` matches what the official client sends.

Probes, in order: baseline 200; the three new fields together 200; the
inline tool_choice 200; the nested form 422.

Also lands the adapter-level regression test for the mid-stream socket
reset fixed in the previous commit — it drives a real reset through
buildGrokFetch and the SSE transform, so it needs both changes present.
This commit is contained in:
程序员阿江(Relakkes)
2026-07-27 02:13:11 +08:00
parent c4560e76d0
commit cbef0360e2
4 changed files with 300 additions and 14 deletions
@@ -621,6 +621,66 @@ describe('anthropicToOpenaiResponses', () => {
expect((result as Record<string, unknown>).stop).toBeUndefined()
expect((result as Record<string, unknown>).stop_sequences).toBeUndefined()
})
// Responses names the function inline. The nested {function:{name}} shape is
// Chat Completions syntax and strict upstreams (xAI) reject it.
test('tool_choice type=tool names the function inline, not nested', () => {
const req: AnthropicRequest = {
model: 'gpt-4o',
max_tokens: 100,
messages: [{ role: 'user', content: 'Hi' }],
tools: [{ name: 'get_weather', input_schema: { type: 'object' } }],
tool_choice: { type: 'tool', name: 'get_weather' },
}
const result = anthropicToOpenaiResponses(req)
expect(result.tool_choice).toEqual({ type: 'function', name: 'get_weather' })
})
test('drops tool_choice when the request carries no tools', () => {
for (const choice of [
{ type: 'auto' },
{ type: 'any' },
{ type: 'tool', name: 'get_weather' },
]) {
const result = anthropicToOpenaiResponses({
model: 'gpt-4o',
max_tokens: 100,
messages: [{ role: 'user', content: 'Hi' }],
tool_choice: choice,
} as AnthropicRequest)
expect(result.tools).toBeUndefined()
expect(result.tool_choice).toBeUndefined()
}
})
test('drops a tool_choice orphaned by tool filtering', () => {
const result = anthropicToOpenaiResponses({
model: 'gpt-4o',
max_tokens: 100,
messages: [{ role: 'user', content: 'Hi' }],
// BatchTool is filtered out of the tool list, so a choice naming it
// would point at a tool the upstream never receives.
tools: [{ name: 'BatchTool', input_schema: { type: 'object' } }],
tool_choice: { type: 'tool', name: 'BatchTool' },
} as AnthropicRequest)
expect(result.tools).toBeUndefined()
expect(result.tool_choice).toBeUndefined()
})
test('keeps a tool_choice whose target survives filtering', () => {
const result = anthropicToOpenaiResponses({
model: 'gpt-4o',
max_tokens: 100,
messages: [{ role: 'user', content: 'Hi' }],
tools: [
{ name: 'BatchTool', input_schema: { type: 'object' } },
{ name: 'get_weather', input_schema: { type: 'object' } },
],
tool_choice: { type: 'tool', name: 'get_weather' },
} as AnthropicRequest)
expect(result.tools).toHaveLength(1)
expect(result.tool_choice).toEqual({ type: 'function', name: 'get_weather' })
})
})
// ─── openaiResponsesToAnthropic ─────────────────────────────────
@@ -68,9 +68,10 @@ export function anthropicToOpenaiResponses(
if (body.top_p !== undefined) result.top_p = body.top_p
}
// tools
// tools — an empty array after filtering is dropped, not sent as `[]`, so it
// reads as "no tools" to strict upstreams instead of "an empty tool set".
if (body.tools && body.tools.length > 0) {
result.tools = body.tools
const tools = body.tools
.filter((t) => t.name !== 'BatchTool')
.map((t) => ({
type: 'function',
@@ -78,11 +79,20 @@ export function anthropicToOpenaiResponses(
description: t.description,
parameters: t.input_schema,
}))
if (tools.length > 0) {
result.tools = tools
}
}
// tool_choice
// tool_choice — only meaningful next to the tools it selects from. A choice
// that outlives its tool (BatchTool filtered above, or a client that sends
// tool_choice with no tools at all) is an orphan that strict Responses
// upstreams reject.
if (body.tool_choice !== undefined) {
result.tool_choice = convertToolChoice(body.tool_choice)
const toolChoice = convertToolChoice(body.tool_choice)
if (isSelectableToolChoice(toolChoice, result.tools)) {
result.tool_choice = toolChoice
}
}
// thinking → reasoning
@@ -184,8 +194,27 @@ function convertToolChoice(choice: unknown): unknown {
if (c.type === 'any') return 'required'
if (c.type === 'none') return 'none'
if (c.type === 'tool' && typeof c.name === 'string') {
return { type: 'function', function: { name: c.name } }
// Responses names the function inline: {type:'function', name}. The
// nested {function:{name}} form belongs to Chat Completions and is
// rejected here (see anthropicToOpenaiChat for that shape).
return { type: 'function', name: c.name }
}
}
return 'auto'
}
/**
* A named tool_choice is only valid while its target survives into the request.
* Anything else — a choice with no tools at all — is dropped so the upstream
* never sees a selector pointing at nothing.
*/
function isSelectableToolChoice(
choice: unknown,
tools: { name: string }[] | undefined,
): boolean {
if (!tools || tools.length === 0) return false
if (typeof choice !== 'object' || choice === null) return true
const name = (choice as Record<string, unknown>).name
if (typeof name !== 'string') return true
return tools.some((tool) => tool.name === name)
}
+167
View File
@@ -9,6 +9,7 @@ import {
} from './fetch.js'
import { GROK_OAUTH_FILE_ENV_KEY } from './storage.js'
import { GROK_OAUTH_TOKEN_ENDPOINT } from './client.js'
import { isRetryableStreamTransportError } from '../api/withRetry.js'
describe('Grok Responses fetch adapter', () => {
let tmpDir: string
@@ -66,6 +67,7 @@ describe('Grok Responses fetch adapter', () => {
expect(call?.headers.get('Authorization')).toBe('Bearer access')
expect(call?.headers.get('X-XAI-Token-Auth')).toBe('xai-grok-cli')
expect(call?.headers.get('x-grok-client-version')).toBe(GROK_CLI_VERSION)
expect(call?.headers.get('x-grok-client-mode')).toBe('interactive')
expect(call?.headers.get('User-Agent')).toBe(`xai-grok-workspace/${GROK_CLI_VERSION}`)
expect(call?.headers.get('x-grok-model-override')).toBe('grok-4.5')
expect(call?.body.model).toBe('grok-4.5')
@@ -167,6 +169,171 @@ describe('Grok Responses fetch adapter', () => {
})
})
// Every turn of one conversation must land on the same cache entry, or the
// whole prefix is re-billed each time. xAI reads the identity from the body,
// the CLI proxy from a header — both must carry it, and both must survive the
// 401 refresh retry.
describe('prompt cache identity', () => {
async function capture(
body: Record<string, unknown>,
init: RequestInit = {},
): Promise<{ headers: Headers; body: Record<string, unknown> }[]> {
const calls: { headers: Headers; body: Record<string, unknown> }[] = []
const fetchOverride: typeof fetch = async (input, requestInit) => {
if (String(input) === GROK_OAUTH_TOKEN_ENDPOINT) {
return Response.json({
access_token: 'new-access',
refresh_token: 'new-refresh',
expires_in: 3600,
})
}
calls.push({
headers: new Headers(requestInit?.headers),
body: JSON.parse(String(requestInit?.body)),
})
return new Response(
'event: response.completed\ndata: {"response":{"id":"r","object":"response","created_at":1,"model":"grok-4.5","status":"completed","output":[],"usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}\n\n',
{ headers: { 'Content-Type': 'text/event-stream' } },
)
}
await buildGrokFetch(fetchOverride, 'test')(
'https://api.anthropic.com/v1/messages',
{ method: 'POST', ...init, body: JSON.stringify(body) },
)
return calls
}
const anthropicBody = (metadata?: Record<string, unknown>) => ({
model: 'grok-4.5',
max_tokens: 64,
messages: [{ role: 'user', content: 'hello' }],
...(metadata ? { metadata } : {}),
})
test("derives it from Claude Code's session-suffixed user_id", async () => {
const calls = await capture(
anthropicBody({ user_id: 'user_abc_session_sess-123' }),
)
expect(calls[0]?.body.prompt_cache_key).toBe('sess-123')
expect(calls[0]?.headers.get('x-grok-conv-id')).toBe('sess-123')
})
test('falls back to the CLI session header', async () => {
const calls = await capture(anthropicBody(), {
headers: { 'X-Claude-Code-Session-Id': 'sess-header' },
})
expect(calls[0]?.body.prompt_cache_key).toBe('sess-header')
expect(calls[0]?.headers.get('x-grok-conv-id')).toBe('sess-header')
})
test('sends no identity rather than an unstable one', async () => {
const calls = await capture(anthropicBody())
expect(calls[0]?.body.prompt_cache_key).toBeUndefined()
expect(calls[0]?.headers.get('x-grok-conv-id')).toBeNull()
})
test('keeps the identity on the retry after a 401 refresh', async () => {
const calls: { headers: Headers; body: Record<string, unknown> }[] = []
const fetchOverride: typeof fetch = async (input, requestInit) => {
if (String(input) === GROK_OAUTH_TOKEN_ENDPOINT) {
return Response.json({
access_token: 'new-access',
refresh_token: 'new-refresh',
expires_in: 3600,
})
}
calls.push({
headers: new Headers(requestInit?.headers),
body: JSON.parse(String(requestInit?.body)),
})
if (calls.length === 1) return new Response('expired', { status: 401 })
return new Response(
'event: response.completed\ndata: {"response":{"id":"r","object":"response","created_at":1,"model":"grok-4.5","status":"completed","output":[],"usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}\n\n',
{ headers: { 'Content-Type': 'text/event-stream' } },
)
}
await buildGrokFetch(fetchOverride, 'test')(
'https://api.anthropic.com/v1/messages',
{
method: 'POST',
body: JSON.stringify(
anthropicBody({ user_id: 'user_abc_session_sess-retry' }),
),
},
)
expect(calls).toHaveLength(2)
for (const call of calls) {
expect(call.headers.get('x-grok-conv-id')).toBe('sess-retry')
expect(call.headers.get('x-grok-model-override')).toBe('grok-4.5')
expect(call.headers.get('x-grok-client-mode')).toBe('interactive')
expect(call.body.prompt_cache_key).toBe('sess-retry')
}
})
})
// The Grok subscription endpoint is reached over a long-lived TLS stream held
// open by the CLI process, so a proxy/NAT/edge reset lands mid-response
// rather than on stream creation. Drive a real socket reset — not a fake
// error object — so the classifier is pinned to what the runtime actually
// throws through the SSE transform.
test('surfaces a mid-stream socket reset as a retryable transport error', async () => {
const upstream = Bun.listen({
hostname: '127.0.0.1',
port: 0,
socket: {
data(socket) {
socket.write(
'HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nTransfer-Encoding: chunked\r\n\r\n',
)
const event = [
'event: response.created',
'data: {"model":"grok-4.5"}',
'',
'',
].join('\n')
socket.write(`${event.length.toString(16)}\r\n${event}\r\n`)
setTimeout(() => socket.terminate(), 20)
},
},
})
try {
const response = await buildGrokFetch(
(_input, init) => fetch(`http://127.0.0.1:${upstream.port}/v1/responses`, init),
'test',
)('https://api.anthropic.com/v1/messages', {
method: 'POST',
body: JSON.stringify({
model: 'grok-4.5',
max_tokens: 64,
stream: true,
messages: [{ role: 'user', content: 'hello' }],
}),
})
expect(response.status).toBe(200)
let thrown: unknown
try {
const reader = response.body!.getReader()
for (;;) {
const { done } = await reader.read()
if (done) break
}
} catch (error) {
thrown = error
}
// The transform must propagate the fault, not close the stream cleanly —
// a clean close would look like a finished (truncated) turn.
expect(thrown).toBeDefined()
expect(isRetryableStreamTransportError(thrown)).toBe(true)
} finally {
upstream.stop(true)
}
})
test('does not refresh-loop on entitlement failures', async () => {
let calls = 0
const response = await buildGrokFetch(async () => {
+39 -9
View File
@@ -1,3 +1,4 @@
import { resolvePromptCacheKey } from '../../server/proxy/promptCacheKey.js'
import { anthropicToOpenaiResponses } from '../../server/proxy/transform/anthropicToOpenaiResponses.js'
import { openaiResponsesStreamToAnthropic } from '../../server/proxy/streaming/openaiResponsesStreamToAnthropic.js'
import { openaiResponsesStreamToAnthropicResponse } from '../../server/proxy/streaming/openaiResponsesStreamToAnthropicResponse.js'
@@ -22,6 +23,9 @@ export function buildGrokIdentityHeaders(accessToken: string): Headers {
'Content-Type': 'application/json',
'X-XAI-Token-Auth': 'xai-grok-cli',
'x-grok-client-version': GROK_CLI_VERSION,
// The CLI gateway distinguishes interactive sessions from batch traffic;
// the official client always declares one.
'x-grok-client-mode': 'interactive',
'User-Agent': `xai-grok-workspace/${GROK_CLI_VERSION}`,
})
}
@@ -38,10 +42,14 @@ export function buildGrokFetch(
const originalBody = await readAnthropicBody(input, init)
const requestedModel = resolveGrokModel(originalBody.model)
const transformedBody = anthropicToOpenaiResponses({
...originalBody,
model: requestedModel,
})
// One conversation must route to one cache entry, or every turn re-bills
// the whole prefix. xAI reads it from the body; the CLI proxy also keys its
// conversation state off a header, so both carry the same identity.
const cacheKey = resolvePromptCacheKey(originalBody, readSessionId(input, init))
const transformedBody = anthropicToOpenaiResponses(
{ ...originalBody, model: requestedModel },
cacheKey ? { cacheKey } : {},
)
transformedBody.model = requestedModel
transformedBody.stream = true
if (grokModelRejectsReasoningEffort(requestedModel)) {
@@ -54,8 +62,14 @@ export function buildGrokFetch(
'Grok OAuth token is missing or expired. Authorize Grok again in the desktop app.',
)
}
const headers = buildGrokIdentityHeaders(tokens.accessToken)
headers.set('x-grok-model-override', requestedModel)
const applyRequestIdentity = (requestHeaders: Headers): Headers => {
requestHeaders.set('x-grok-model-override', requestedModel)
if (cacheKey) requestHeaders.set('x-grok-conv-id', cacheKey)
return requestHeaders
}
const headers = applyRequestIdentity(
buildGrokIdentityHeaders(tokens.accessToken),
)
void source
const requestUpstream = (requestHeaders: Headers) => inner(GROK_CLI_API_ENDPOINT, {
@@ -69,9 +83,9 @@ export function buildGrokFetch(
if (upstream.status === 401) {
const refreshed = await forceRefreshGrokTokens({ fetchOverride: inner })
if (refreshed) {
const refreshedHeaders = buildGrokIdentityHeaders(refreshed.accessToken)
refreshedHeaders.set('x-grok-model-override', requestedModel)
upstream = await requestUpstream(refreshedHeaders)
upstream = await requestUpstream(
applyRequestIdentity(buildGrokIdentityHeaders(refreshed.accessToken)),
)
}
}
@@ -118,6 +132,22 @@ export function buildGrokFetch(
}
}
/**
* The CLI stamps its session id on every request as a default header, which is
* the last-resort cache identity when the body carries no session metadata.
*/
function readSessionId(
input: RequestInfo | URL,
init?: RequestInit,
): string | null {
const header = 'x-claude-code-session-id'
if (init?.headers) {
const fromInit = new Headers(init.headers).get(header)
if (fromInit) return fromInit
}
return input instanceof Request ? input.headers.get(header) : null
}
async function readAnthropicBody(
input: RequestInfo | URL,
init?: RequestInit,