mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 20:03:13 +08:00
fix(grok): align the Responses request with what xAI actually accepts
Four gaps against the official Grok CLI surface, found by comparing this
adapter with sub2api's and each confirmed by a probe against the live
endpoint.
The first is not a parity gap but a hard failure. A named tool_choice was
translated into Chat Completions syntax, `{type:'function',function:{name}}`,
which the Responses endpoint refuses:
422 data did not match any variant of untagged enum ModelToolChoice
so every request where the caller pins a tool failed outright. Responses
names the function inline. Only the Responses transform changes; the Chat
transform keeps the nested form, which is correct there.
The rest:
- A tool_choice sent with no tools, or one whose target the BatchTool
filter removed, is dropped rather than left selecting a tool the upstream
never receives — and a tool list that filters down to empty is omitted
instead of sent as `[]`.
- `prompt_cache_key` and the `x-grok-conv-id` header carry one identity per
conversation, reusing resolvePromptCacheKey (the Claude Code session id)
so a turn stops re-billing the whole prefix. Both survive the 401 refresh
retry: header assembly is folded into one helper so the first attempt and
the retry cannot drift apart.
- `x-grok-client-mode: interactive` matches what the official client sends.
Probes, in order: baseline 200; the three new fields together 200; the
inline tool_choice 200; the nested form 422.
Also lands the adapter-level regression test for the mid-stream socket
reset fixed in the previous commit — it drives a real reset through
buildGrokFetch and the SSE transform, so it needs both changes present.
This commit is contained in:
@@ -621,6 +621,66 @@ describe('anthropicToOpenaiResponses', () => {
|
||||
expect((result as Record<string, unknown>).stop).toBeUndefined()
|
||||
expect((result as Record<string, unknown>).stop_sequences).toBeUndefined()
|
||||
})
|
||||
|
||||
// Responses names the function inline. The nested {function:{name}} shape is
|
||||
// Chat Completions syntax and strict upstreams (xAI) reject it.
|
||||
test('tool_choice type=tool names the function inline, not nested', () => {
|
||||
const req: AnthropicRequest = {
|
||||
model: 'gpt-4o',
|
||||
max_tokens: 100,
|
||||
messages: [{ role: 'user', content: 'Hi' }],
|
||||
tools: [{ name: 'get_weather', input_schema: { type: 'object' } }],
|
||||
tool_choice: { type: 'tool', name: 'get_weather' },
|
||||
}
|
||||
const result = anthropicToOpenaiResponses(req)
|
||||
expect(result.tool_choice).toEqual({ type: 'function', name: 'get_weather' })
|
||||
})
|
||||
|
||||
test('drops tool_choice when the request carries no tools', () => {
|
||||
for (const choice of [
|
||||
{ type: 'auto' },
|
||||
{ type: 'any' },
|
||||
{ type: 'tool', name: 'get_weather' },
|
||||
]) {
|
||||
const result = anthropicToOpenaiResponses({
|
||||
model: 'gpt-4o',
|
||||
max_tokens: 100,
|
||||
messages: [{ role: 'user', content: 'Hi' }],
|
||||
tool_choice: choice,
|
||||
} as AnthropicRequest)
|
||||
expect(result.tools).toBeUndefined()
|
||||
expect(result.tool_choice).toBeUndefined()
|
||||
}
|
||||
})
|
||||
|
||||
test('drops a tool_choice orphaned by tool filtering', () => {
|
||||
const result = anthropicToOpenaiResponses({
|
||||
model: 'gpt-4o',
|
||||
max_tokens: 100,
|
||||
messages: [{ role: 'user', content: 'Hi' }],
|
||||
// BatchTool is filtered out of the tool list, so a choice naming it
|
||||
// would point at a tool the upstream never receives.
|
||||
tools: [{ name: 'BatchTool', input_schema: { type: 'object' } }],
|
||||
tool_choice: { type: 'tool', name: 'BatchTool' },
|
||||
} as AnthropicRequest)
|
||||
expect(result.tools).toBeUndefined()
|
||||
expect(result.tool_choice).toBeUndefined()
|
||||
})
|
||||
|
||||
test('keeps a tool_choice whose target survives filtering', () => {
|
||||
const result = anthropicToOpenaiResponses({
|
||||
model: 'gpt-4o',
|
||||
max_tokens: 100,
|
||||
messages: [{ role: 'user', content: 'Hi' }],
|
||||
tools: [
|
||||
{ name: 'BatchTool', input_schema: { type: 'object' } },
|
||||
{ name: 'get_weather', input_schema: { type: 'object' } },
|
||||
],
|
||||
tool_choice: { type: 'tool', name: 'get_weather' },
|
||||
} as AnthropicRequest)
|
||||
expect(result.tools).toHaveLength(1)
|
||||
expect(result.tool_choice).toEqual({ type: 'function', name: 'get_weather' })
|
||||
})
|
||||
})
|
||||
|
||||
// ─── openaiResponsesToAnthropic ─────────────────────────────────
|
||||
|
||||
@@ -68,9 +68,10 @@ export function anthropicToOpenaiResponses(
|
||||
if (body.top_p !== undefined) result.top_p = body.top_p
|
||||
}
|
||||
|
||||
// tools
|
||||
// tools — an empty array after filtering is dropped, not sent as `[]`, so it
|
||||
// reads as "no tools" to strict upstreams instead of "an empty tool set".
|
||||
if (body.tools && body.tools.length > 0) {
|
||||
result.tools = body.tools
|
||||
const tools = body.tools
|
||||
.filter((t) => t.name !== 'BatchTool')
|
||||
.map((t) => ({
|
||||
type: 'function',
|
||||
@@ -78,11 +79,20 @@ export function anthropicToOpenaiResponses(
|
||||
description: t.description,
|
||||
parameters: t.input_schema,
|
||||
}))
|
||||
if (tools.length > 0) {
|
||||
result.tools = tools
|
||||
}
|
||||
}
|
||||
|
||||
// tool_choice
|
||||
// tool_choice — only meaningful next to the tools it selects from. A choice
|
||||
// that outlives its tool (BatchTool filtered above, or a client that sends
|
||||
// tool_choice with no tools at all) is an orphan that strict Responses
|
||||
// upstreams reject.
|
||||
if (body.tool_choice !== undefined) {
|
||||
result.tool_choice = convertToolChoice(body.tool_choice)
|
||||
const toolChoice = convertToolChoice(body.tool_choice)
|
||||
if (isSelectableToolChoice(toolChoice, result.tools)) {
|
||||
result.tool_choice = toolChoice
|
||||
}
|
||||
}
|
||||
|
||||
// thinking → reasoning
|
||||
@@ -184,8 +194,27 @@ function convertToolChoice(choice: unknown): unknown {
|
||||
if (c.type === 'any') return 'required'
|
||||
if (c.type === 'none') return 'none'
|
||||
if (c.type === 'tool' && typeof c.name === 'string') {
|
||||
return { type: 'function', function: { name: c.name } }
|
||||
// Responses names the function inline: {type:'function', name}. The
|
||||
// nested {function:{name}} form belongs to Chat Completions and is
|
||||
// rejected here (see anthropicToOpenaiChat for that shape).
|
||||
return { type: 'function', name: c.name }
|
||||
}
|
||||
}
|
||||
return 'auto'
|
||||
}
|
||||
|
||||
/**
|
||||
* A named tool_choice is only valid while its target survives into the request.
|
||||
* Anything else — a choice with no tools at all — is dropped so the upstream
|
||||
* never sees a selector pointing at nothing.
|
||||
*/
|
||||
function isSelectableToolChoice(
|
||||
choice: unknown,
|
||||
tools: { name: string }[] | undefined,
|
||||
): boolean {
|
||||
if (!tools || tools.length === 0) return false
|
||||
if (typeof choice !== 'object' || choice === null) return true
|
||||
const name = (choice as Record<string, unknown>).name
|
||||
if (typeof name !== 'string') return true
|
||||
return tools.some((tool) => tool.name === name)
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@ import {
|
||||
} from './fetch.js'
|
||||
import { GROK_OAUTH_FILE_ENV_KEY } from './storage.js'
|
||||
import { GROK_OAUTH_TOKEN_ENDPOINT } from './client.js'
|
||||
import { isRetryableStreamTransportError } from '../api/withRetry.js'
|
||||
|
||||
describe('Grok Responses fetch adapter', () => {
|
||||
let tmpDir: string
|
||||
@@ -66,6 +67,7 @@ describe('Grok Responses fetch adapter', () => {
|
||||
expect(call?.headers.get('Authorization')).toBe('Bearer access')
|
||||
expect(call?.headers.get('X-XAI-Token-Auth')).toBe('xai-grok-cli')
|
||||
expect(call?.headers.get('x-grok-client-version')).toBe(GROK_CLI_VERSION)
|
||||
expect(call?.headers.get('x-grok-client-mode')).toBe('interactive')
|
||||
expect(call?.headers.get('User-Agent')).toBe(`xai-grok-workspace/${GROK_CLI_VERSION}`)
|
||||
expect(call?.headers.get('x-grok-model-override')).toBe('grok-4.5')
|
||||
expect(call?.body.model).toBe('grok-4.5')
|
||||
@@ -167,6 +169,171 @@ describe('Grok Responses fetch adapter', () => {
|
||||
})
|
||||
})
|
||||
|
||||
// Every turn of one conversation must land on the same cache entry, or the
|
||||
// whole prefix is re-billed each time. xAI reads the identity from the body,
|
||||
// the CLI proxy from a header — both must carry it, and both must survive the
|
||||
// 401 refresh retry.
|
||||
describe('prompt cache identity', () => {
|
||||
async function capture(
|
||||
body: Record<string, unknown>,
|
||||
init: RequestInit = {},
|
||||
): Promise<{ headers: Headers; body: Record<string, unknown> }[]> {
|
||||
const calls: { headers: Headers; body: Record<string, unknown> }[] = []
|
||||
const fetchOverride: typeof fetch = async (input, requestInit) => {
|
||||
if (String(input) === GROK_OAUTH_TOKEN_ENDPOINT) {
|
||||
return Response.json({
|
||||
access_token: 'new-access',
|
||||
refresh_token: 'new-refresh',
|
||||
expires_in: 3600,
|
||||
})
|
||||
}
|
||||
calls.push({
|
||||
headers: new Headers(requestInit?.headers),
|
||||
body: JSON.parse(String(requestInit?.body)),
|
||||
})
|
||||
return new Response(
|
||||
'event: response.completed\ndata: {"response":{"id":"r","object":"response","created_at":1,"model":"grok-4.5","status":"completed","output":[],"usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}\n\n',
|
||||
{ headers: { 'Content-Type': 'text/event-stream' } },
|
||||
)
|
||||
}
|
||||
await buildGrokFetch(fetchOverride, 'test')(
|
||||
'https://api.anthropic.com/v1/messages',
|
||||
{ method: 'POST', ...init, body: JSON.stringify(body) },
|
||||
)
|
||||
return calls
|
||||
}
|
||||
|
||||
const anthropicBody = (metadata?: Record<string, unknown>) => ({
|
||||
model: 'grok-4.5',
|
||||
max_tokens: 64,
|
||||
messages: [{ role: 'user', content: 'hello' }],
|
||||
...(metadata ? { metadata } : {}),
|
||||
})
|
||||
|
||||
test("derives it from Claude Code's session-suffixed user_id", async () => {
|
||||
const calls = await capture(
|
||||
anthropicBody({ user_id: 'user_abc_session_sess-123' }),
|
||||
)
|
||||
expect(calls[0]?.body.prompt_cache_key).toBe('sess-123')
|
||||
expect(calls[0]?.headers.get('x-grok-conv-id')).toBe('sess-123')
|
||||
})
|
||||
|
||||
test('falls back to the CLI session header', async () => {
|
||||
const calls = await capture(anthropicBody(), {
|
||||
headers: { 'X-Claude-Code-Session-Id': 'sess-header' },
|
||||
})
|
||||
expect(calls[0]?.body.prompt_cache_key).toBe('sess-header')
|
||||
expect(calls[0]?.headers.get('x-grok-conv-id')).toBe('sess-header')
|
||||
})
|
||||
|
||||
test('sends no identity rather than an unstable one', async () => {
|
||||
const calls = await capture(anthropicBody())
|
||||
expect(calls[0]?.body.prompt_cache_key).toBeUndefined()
|
||||
expect(calls[0]?.headers.get('x-grok-conv-id')).toBeNull()
|
||||
})
|
||||
|
||||
test('keeps the identity on the retry after a 401 refresh', async () => {
|
||||
const calls: { headers: Headers; body: Record<string, unknown> }[] = []
|
||||
const fetchOverride: typeof fetch = async (input, requestInit) => {
|
||||
if (String(input) === GROK_OAUTH_TOKEN_ENDPOINT) {
|
||||
return Response.json({
|
||||
access_token: 'new-access',
|
||||
refresh_token: 'new-refresh',
|
||||
expires_in: 3600,
|
||||
})
|
||||
}
|
||||
calls.push({
|
||||
headers: new Headers(requestInit?.headers),
|
||||
body: JSON.parse(String(requestInit?.body)),
|
||||
})
|
||||
if (calls.length === 1) return new Response('expired', { status: 401 })
|
||||
return new Response(
|
||||
'event: response.completed\ndata: {"response":{"id":"r","object":"response","created_at":1,"model":"grok-4.5","status":"completed","output":[],"usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}\n\n',
|
||||
{ headers: { 'Content-Type': 'text/event-stream' } },
|
||||
)
|
||||
}
|
||||
await buildGrokFetch(fetchOverride, 'test')(
|
||||
'https://api.anthropic.com/v1/messages',
|
||||
{
|
||||
method: 'POST',
|
||||
body: JSON.stringify(
|
||||
anthropicBody({ user_id: 'user_abc_session_sess-retry' }),
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
expect(calls).toHaveLength(2)
|
||||
for (const call of calls) {
|
||||
expect(call.headers.get('x-grok-conv-id')).toBe('sess-retry')
|
||||
expect(call.headers.get('x-grok-model-override')).toBe('grok-4.5')
|
||||
expect(call.headers.get('x-grok-client-mode')).toBe('interactive')
|
||||
expect(call.body.prompt_cache_key).toBe('sess-retry')
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
// The Grok subscription endpoint is reached over a long-lived TLS stream held
|
||||
// open by the CLI process, so a proxy/NAT/edge reset lands mid-response
|
||||
// rather than on stream creation. Drive a real socket reset — not a fake
|
||||
// error object — so the classifier is pinned to what the runtime actually
|
||||
// throws through the SSE transform.
|
||||
test('surfaces a mid-stream socket reset as a retryable transport error', async () => {
|
||||
const upstream = Bun.listen({
|
||||
hostname: '127.0.0.1',
|
||||
port: 0,
|
||||
socket: {
|
||||
data(socket) {
|
||||
socket.write(
|
||||
'HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nTransfer-Encoding: chunked\r\n\r\n',
|
||||
)
|
||||
const event = [
|
||||
'event: response.created',
|
||||
'data: {"model":"grok-4.5"}',
|
||||
'',
|
||||
'',
|
||||
].join('\n')
|
||||
socket.write(`${event.length.toString(16)}\r\n${event}\r\n`)
|
||||
setTimeout(() => socket.terminate(), 20)
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
try {
|
||||
const response = await buildGrokFetch(
|
||||
(_input, init) => fetch(`http://127.0.0.1:${upstream.port}/v1/responses`, init),
|
||||
'test',
|
||||
)('https://api.anthropic.com/v1/messages', {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({
|
||||
model: 'grok-4.5',
|
||||
max_tokens: 64,
|
||||
stream: true,
|
||||
messages: [{ role: 'user', content: 'hello' }],
|
||||
}),
|
||||
})
|
||||
|
||||
expect(response.status).toBe(200)
|
||||
|
||||
let thrown: unknown
|
||||
try {
|
||||
const reader = response.body!.getReader()
|
||||
for (;;) {
|
||||
const { done } = await reader.read()
|
||||
if (done) break
|
||||
}
|
||||
} catch (error) {
|
||||
thrown = error
|
||||
}
|
||||
|
||||
// The transform must propagate the fault, not close the stream cleanly —
|
||||
// a clean close would look like a finished (truncated) turn.
|
||||
expect(thrown).toBeDefined()
|
||||
expect(isRetryableStreamTransportError(thrown)).toBe(true)
|
||||
} finally {
|
||||
upstream.stop(true)
|
||||
}
|
||||
})
|
||||
|
||||
test('does not refresh-loop on entitlement failures', async () => {
|
||||
let calls = 0
|
||||
const response = await buildGrokFetch(async () => {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { resolvePromptCacheKey } from '../../server/proxy/promptCacheKey.js'
|
||||
import { anthropicToOpenaiResponses } from '../../server/proxy/transform/anthropicToOpenaiResponses.js'
|
||||
import { openaiResponsesStreamToAnthropic } from '../../server/proxy/streaming/openaiResponsesStreamToAnthropic.js'
|
||||
import { openaiResponsesStreamToAnthropicResponse } from '../../server/proxy/streaming/openaiResponsesStreamToAnthropicResponse.js'
|
||||
@@ -22,6 +23,9 @@ export function buildGrokIdentityHeaders(accessToken: string): Headers {
|
||||
'Content-Type': 'application/json',
|
||||
'X-XAI-Token-Auth': 'xai-grok-cli',
|
||||
'x-grok-client-version': GROK_CLI_VERSION,
|
||||
// The CLI gateway distinguishes interactive sessions from batch traffic;
|
||||
// the official client always declares one.
|
||||
'x-grok-client-mode': 'interactive',
|
||||
'User-Agent': `xai-grok-workspace/${GROK_CLI_VERSION}`,
|
||||
})
|
||||
}
|
||||
@@ -38,10 +42,14 @@ export function buildGrokFetch(
|
||||
|
||||
const originalBody = await readAnthropicBody(input, init)
|
||||
const requestedModel = resolveGrokModel(originalBody.model)
|
||||
const transformedBody = anthropicToOpenaiResponses({
|
||||
...originalBody,
|
||||
model: requestedModel,
|
||||
})
|
||||
// One conversation must route to one cache entry, or every turn re-bills
|
||||
// the whole prefix. xAI reads it from the body; the CLI proxy also keys its
|
||||
// conversation state off a header, so both carry the same identity.
|
||||
const cacheKey = resolvePromptCacheKey(originalBody, readSessionId(input, init))
|
||||
const transformedBody = anthropicToOpenaiResponses(
|
||||
{ ...originalBody, model: requestedModel },
|
||||
cacheKey ? { cacheKey } : {},
|
||||
)
|
||||
transformedBody.model = requestedModel
|
||||
transformedBody.stream = true
|
||||
if (grokModelRejectsReasoningEffort(requestedModel)) {
|
||||
@@ -54,8 +62,14 @@ export function buildGrokFetch(
|
||||
'Grok OAuth token is missing or expired. Authorize Grok again in the desktop app.',
|
||||
)
|
||||
}
|
||||
const headers = buildGrokIdentityHeaders(tokens.accessToken)
|
||||
headers.set('x-grok-model-override', requestedModel)
|
||||
const applyRequestIdentity = (requestHeaders: Headers): Headers => {
|
||||
requestHeaders.set('x-grok-model-override', requestedModel)
|
||||
if (cacheKey) requestHeaders.set('x-grok-conv-id', cacheKey)
|
||||
return requestHeaders
|
||||
}
|
||||
const headers = applyRequestIdentity(
|
||||
buildGrokIdentityHeaders(tokens.accessToken),
|
||||
)
|
||||
|
||||
void source
|
||||
const requestUpstream = (requestHeaders: Headers) => inner(GROK_CLI_API_ENDPOINT, {
|
||||
@@ -69,9 +83,9 @@ export function buildGrokFetch(
|
||||
if (upstream.status === 401) {
|
||||
const refreshed = await forceRefreshGrokTokens({ fetchOverride: inner })
|
||||
if (refreshed) {
|
||||
const refreshedHeaders = buildGrokIdentityHeaders(refreshed.accessToken)
|
||||
refreshedHeaders.set('x-grok-model-override', requestedModel)
|
||||
upstream = await requestUpstream(refreshedHeaders)
|
||||
upstream = await requestUpstream(
|
||||
applyRequestIdentity(buildGrokIdentityHeaders(refreshed.accessToken)),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -118,6 +132,22 @@ export function buildGrokFetch(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The CLI stamps its session id on every request as a default header, which is
|
||||
* the last-resort cache identity when the body carries no session metadata.
|
||||
*/
|
||||
function readSessionId(
|
||||
input: RequestInfo | URL,
|
||||
init?: RequestInit,
|
||||
): string | null {
|
||||
const header = 'x-claude-code-session-id'
|
||||
if (init?.headers) {
|
||||
const fromInit = new Headers(init.headers).get(header)
|
||||
if (fromInit) return fromInit
|
||||
}
|
||||
return input instanceof Request ? input.headers.get(header) : null
|
||||
}
|
||||
|
||||
async function readAnthropicBody(
|
||||
input: RequestInfo | URL,
|
||||
init?: RequestInit,
|
||||
|
||||
Reference in New Issue
Block a user